Skip to content

API reference

Use the guides for workflows and examples. This reference follows the public import paths used by applications.

Resources and clients

Resource

Bases: BaseResource

Source code in cloudcoil/resources.py
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
class Resource(BaseResource):
    # Generated from API paths/CRDs, so manifest generation needs no discovery.
    __cloudcoil_api__: ClassVar[dict[str, Any] | None] = None
    metadata: ObjectMeta | None = None

    @classmethod
    def client(
        cls,
        config: "Config | None" = None,
        *,
        namespace: str | None = None,
        cached: bool | None = None,
    ) -> "APIClient[Self]":
        """Get this resource's typed client using an explicit or active Config.

        The returned client shares Config's transport and lifetime. A namespace
        override applies only to this client; it does not change Config.
        """
        config = config if config is not None else context.active_config
        client = config.client_for(cls, sync=True, cached=cached)
        if namespace is not None:
            client.default_namespace = namespace
        return client

    @classmethod
    async def async_client(
        cls,
        config: "Config | None" = None,
        *,
        namespace: str | None = None,
        cached: bool | None = None,
    ) -> "AsyncAPIClient[Self]":
        """Get a typed client, discovering resources without blocking the loop."""
        config = config if config is not None else context.active_config
        client = await config.async_client_for(cls, cached=cached)
        if namespace is not None:
            client.default_namespace = namespace
        return client

    @classmethod
    def from_file(cls, path: str | Path) -> Self:
        path = Path(path)
        return cls.model_validate(yaml.safe_load(path.read_text()))

    @property
    def name(self) -> str | None:
        if self.metadata is None:
            return None
        return self.metadata.name

    @name.setter
    def name(self, value: str):
        if self.metadata is None:
            self.metadata = ObjectMeta(name=value)
        else:
            self.metadata.name = value

    @property
    def namespace(self) -> str | None:
        if self.metadata is None:
            return None
        return self.metadata.namespace

    @namespace.setter
    def namespace(self, value: str):
        if self.metadata is None:
            self.metadata = ObjectMeta(namespace=value)
        else:
            self.metadata.namespace = value

    @property
    def resource_version(self) -> str | None:
        if self.metadata is None:
            return None
        return self.metadata.resource_version

    @classmethod
    def get(cls, name: str, namespace: str | None = None) -> Self:
        config = context.active_config
        return config.client_for(cls, sync=True).get(name, namespace)

    @classmethod
    async def async_get(cls, name: str, namespace: str | None = None) -> Self:
        config = context.active_config
        return await (await config.async_client_for(cls)).get(name, namespace)

    def fetch(self) -> Self:
        config = context.active_config
        if self.name is None:
            raise ValueError("Resource name is not set")
        return config.client_for(self.__class__, sync=True).get(self.name, self.namespace)

    async def async_fetch(self) -> Self:
        config = context.active_config
        if self.name is None:
            raise ValueError("Resource name is not set")
        return await (await config.async_client_for(self.__class__)).get(self.name, self.namespace)

    def create(self, dry_run: bool = False) -> Self:
        config = context.active_config
        return config.client_for(self.__class__, sync=True).create(self, dry_run=dry_run)

    async def async_create(self, dry_run: bool = False) -> Self:
        config = context.active_config
        return await (await config.async_client_for(self.__class__)).create(self, dry_run=dry_run)

    def update(self, dry_run: bool = False) -> Self:
        config = context.active_config
        return config.client_for(self.__class__, sync=True).update(self, dry_run=dry_run)

    async def async_update(self, dry_run: bool = False) -> Self:
        config = context.active_config
        return await (await config.async_client_for(self.__class__)).update(self, dry_run=dry_run)

    def patch(
        self,
        operations: list[dict[str, Any]],
        *,
        subresource: Literal["status"] | None = None,
        dry_run: bool = False,
    ) -> Self:
        """Apply a JSON Patch; cloudcoil.patches.diff adds UID/version preconditions."""
        config = context.active_config
        return config.client_for(self.__class__, sync=True).patch(
            self, operations, subresource=subresource, dry_run=dry_run
        )

    async def async_patch(
        self,
        operations: list[dict[str, Any]],
        *,
        subresource: Literal["status"] | None = None,
        dry_run: bool = False,
    ) -> Self:
        """Asynchronously apply a JSON Patch to this resource or its status."""
        config = context.active_config
        return await (await config.async_client_for(self.__class__)).patch(
            self, operations, subresource=subresource, dry_run=dry_run
        )

    def update_status(self, dry_run: bool = False) -> Self:
        config = context.active_config
        return config.client_for(self.__class__, sync=True).update_status(self, dry_run=dry_run)

    async def async_update_status(self, dry_run: bool = False) -> Self:
        config = context.active_config
        return await (await config.async_client_for(self.__class__)).update_status(
            self, dry_run=dry_run
        )

    def save(self, dry_run: bool = False) -> Self:
        """Create or replace, preserving optimistic concurrency on existing resources."""
        if self.name is None:
            return self.create(dry_run=dry_run)
        if self.resource_version is not None:
            return self.update(dry_run=dry_run)
        try:
            existing = self.fetch()
        except ResourceNotFound:
            return self.create(dry_run=dry_run)
        body = self.model_copy(deep=True)
        assert body.metadata is not None
        body.metadata.resource_version = existing.resource_version
        return body.update(dry_run=dry_run)

    async def async_save(self, dry_run: bool = False) -> Self:
        if self.name is None:
            return await self.async_create(dry_run=dry_run)
        if self.resource_version is not None:
            return await self.async_update(dry_run=dry_run)
        try:
            existing = await self.async_fetch()
        except ResourceNotFound:
            return await self.async_create(dry_run=dry_run)
        body = self.model_copy(deep=True)
        assert body.metadata is not None
        body.metadata.resource_version = existing.resource_version
        return await body.async_update(dry_run=dry_run)

    @classmethod
    def delete(
        cls,
        name: str,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
    ) -> Self | Status:
        config = context.active_config
        return config.client_for(cls, sync=True).delete(
            name,
            namespace,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
        )

    @classmethod
    async def async_delete(
        cls,
        name: str,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
    ) -> Self | Status:
        config = context.active_config
        return await (await config.async_client_for(cls)).delete(
            name,
            namespace,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
        )

    def remove(
        self,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
    ) -> Self | Status:
        config = context.active_config
        return config.client_for(self.__class__, sync=True).remove(
            self,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
        )

    async def async_remove(
        self,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
    ) -> Self | Status:
        config = context.active_config
        return await (await config.async_client_for(self.__class__)).remove(
            self,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
        )

    @classmethod
    def list(
        cls,
        namespace: str | None = None,
        all_namespaces: bool = False,
        continue_: None | str = None,
        field_selector: str | None = None,
        label_selector: str | None = None,
        limit: int = DEFAULT_PAGE_LIMIT,
    ) -> "ResourceList[Self]":
        config = context.active_config
        return config.client_for(cls, sync=True).list(
            namespace=namespace,
            all_namespaces=all_namespaces,
            continue_=continue_,
            field_selector=field_selector,
            label_selector=label_selector,
            limit=limit,
        )

    @classmethod
    async def async_list(
        cls,
        namespace: str | None = None,
        all_namespaces: bool = False,
        continue_: None | str = None,
        field_selector: str | None = None,
        label_selector: str | None = None,
        limit: int = DEFAULT_PAGE_LIMIT,
    ) -> "ResourceList[Self]":
        config = context.active_config
        return await (await config.async_client_for(cls)).list(
            namespace=namespace,
            all_namespaces=all_namespaces,
            continue_=continue_,
            field_selector=field_selector,
            label_selector=label_selector,
            limit=limit,
        )

    @classmethod
    def delete_all(
        cls,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
        label_selector: str | None = None,
        field_selector: str | None = None,
    ) -> "ResourceList[Self]":
        config = context.active_config
        return config.client_for(cls, sync=True).delete_all(
            namespace=namespace,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
            label_selector=label_selector,
            field_selector=field_selector,
        )

    @classmethod
    async def async_delete_all(
        cls,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
        label_selector: str | None = None,
        field_selector: str | None = None,
    ) -> "ResourceList[Self]":
        config = context.active_config
        return await (await config.async_client_for(cls)).delete_all(
            namespace=namespace,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
            label_selector=label_selector,
            field_selector=field_selector,
        )

    @classmethod
    def watch(
        cls,
        namespace: str | None = None,
        all_namespaces: bool = False,
        field_selector: str | None = None,
        label_selector: str | None = None,
        resource_version: str | None = None,
    ) -> Iterator[tuple[WatchEvent, "Self"] | tuple[BookmarkEvent, "Unstructured"]]:
        config = context.active_config
        client_result = config.client_for(cls, sync=True).watch(
            namespace=namespace,
            all_namespaces=all_namespaces,
            field_selector=field_selector,
            label_selector=label_selector,
            resource_version=resource_version,
        )
        # Type cast to ensure proper return type is recognized
        return client_result  # type: ignore

    @classmethod
    async def async_watch(
        cls,
        namespace: str | None = None,
        all_namespaces: bool = False,
        field_selector: str | None = None,
        label_selector: str | None = None,
        resource_version: str | None = None,
    ) -> AsyncGenerator[tuple[WatchEvent, "Self"] | tuple[BookmarkEvent, "Unstructured"], None]:
        config = context.active_config
        client_result = (await config.async_client_for(cls)).watch(
            namespace=namespace,
            all_namespaces=all_namespaces,
            field_selector=field_selector,
            label_selector=label_selector,
            resource_version=resource_version,
        )
        try:
            async for event in client_result:
                yield event
        finally:
            await client_result.aclose()

    @overload
    def wait_for(
        self,
        predicate: WaitPredicate,
        /,
        timeout: float | None = None,
    ) -> None: ...

    @overload
    def wait_for(
        self,
        predicate: dict[
            str,
            WaitPredicate,
        ],
        /,
        timeout: float | None = None,
    ) -> str: ...

    def wait_for(self, predicate, timeout=None):
        config = context.active_config
        client = config.client_for(self.__class__, sync=True)
        if not self.name:
            raise ValueError("Resource name must be set to wait for it")
        if isinstance(predicate, dict):
            return client.wait_for(self, predicate, timeout=timeout)
        assert isinstance(predicate, Callable)
        client.wait_for(self, {"predicate": predicate}, timeout=timeout)
        return None

    @overload
    async def async_wait_for(
        self,
        predicate: WaitPredicate,
        /,
        timeout: float | None = None,
    ) -> None: ...

    @overload
    async def async_wait_for(
        self,
        predicate: dict[
            str,
            WaitPredicate,
        ],
        /,
        timeout: float | None = None,
    ) -> str: ...

    async def async_wait_for(
        self,
        predicate,
        timeout=None,
    ):
        config = context.active_config
        client = await config.async_client_for(self.__class__)
        if not self.name:
            raise ValueError("Resource name must be set to wait for it")
        if isinstance(predicate, dict):
            return await client.wait_for(self, predicate, timeout=timeout)
        assert isinstance(predicate, Callable)
        await client.wait_for(self, {"predicate": predicate}, timeout=timeout)
        return None

    def scale(self, replicas: int) -> Self:
        """Scale the resource to the specified number of replicas.

        Args:
            replicas: The desired number of replicas

        Returns:
            The updated resource after scaling

        Raises:
            ValueError: If the resource does not support scaling or metadata is not set
            ResourceNotFound: If the resource does not exist
            APIError: If the API request fails
        """
        config = context.active_config
        return config.client_for(self.__class__, sync=True).scale(
            self,
            replicas=replicas,
        )

    async def async_scale(self, replicas: int) -> Self:
        """Asynchronously scale the resource to the specified number of replicas.

        Args:
            replicas: The desired number of replicas

        Returns:
            The updated resource after scaling

        Raises:
            ValueError: If the resource does not support scaling or metadata is not set
            ResourceNotFound: If the resource does not exist
            APIError: If the API request fails
        """
        config = context.active_config
        return await (await config.async_client_for(self.__class__)).scale(
            self,
            replicas=replicas,
        )

async_client(config=None, *, namespace=None, cached=None) async classmethod

Get a typed client, discovering resources without blocking the loop.

Source code in cloudcoil/resources.py
103
104
105
106
107
108
109
110
111
112
113
114
115
116
@classmethod
async def async_client(
    cls,
    config: "Config | None" = None,
    *,
    namespace: str | None = None,
    cached: bool | None = None,
) -> "AsyncAPIClient[Self]":
    """Get a typed client, discovering resources without blocking the loop."""
    config = config if config is not None else context.active_config
    client = await config.async_client_for(cls, cached=cached)
    if namespace is not None:
        client.default_namespace = namespace
    return client

async_patch(operations, *, subresource=None, dry_run=False) async

Asynchronously apply a JSON Patch to this resource or its status.

Source code in cloudcoil/resources.py
206
207
208
209
210
211
212
213
214
215
216
217
async def async_patch(
    self,
    operations: list[dict[str, Any]],
    *,
    subresource: Literal["status"] | None = None,
    dry_run: bool = False,
) -> Self:
    """Asynchronously apply a JSON Patch to this resource or its status."""
    config = context.active_config
    return await (await config.async_client_for(self.__class__)).patch(
        self, operations, subresource=subresource, dry_run=dry_run
    )

async_scale(replicas) async

Asynchronously scale the resource to the specified number of replicas.

Parameters:

Name Type Description Default
replicas int

The desired number of replicas

required

Returns:

Type Description
Self

The updated resource after scaling

Raises:

Type Description
ValueError

If the resource does not support scaling or metadata is not set

ResourceNotFound

If the resource does not exist

APIError

If the API request fails

Source code in cloudcoil/resources.py
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
async def async_scale(self, replicas: int) -> Self:
    """Asynchronously scale the resource to the specified number of replicas.

    Args:
        replicas: The desired number of replicas

    Returns:
        The updated resource after scaling

    Raises:
        ValueError: If the resource does not support scaling or metadata is not set
        ResourceNotFound: If the resource does not exist
        APIError: If the API request fails
    """
    config = context.active_config
    return await (await config.async_client_for(self.__class__)).scale(
        self,
        replicas=replicas,
    )

client(config=None, *, namespace=None, cached=None) classmethod

Get this resource's typed client using an explicit or active Config.

The returned client shares Config's transport and lifetime. A namespace override applies only to this client; it does not change Config.

Source code in cloudcoil/resources.py
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
@classmethod
def client(
    cls,
    config: "Config | None" = None,
    *,
    namespace: str | None = None,
    cached: bool | None = None,
) -> "APIClient[Self]":
    """Get this resource's typed client using an explicit or active Config.

    The returned client shares Config's transport and lifetime. A namespace
    override applies only to this client; it does not change Config.
    """
    config = config if config is not None else context.active_config
    client = config.client_for(cls, sync=True, cached=cached)
    if namespace is not None:
        client.default_namespace = namespace
    return client

patch(operations, *, subresource=None, dry_run=False)

Apply a JSON Patch; cloudcoil.patches.diff adds UID/version preconditions.

Source code in cloudcoil/resources.py
193
194
195
196
197
198
199
200
201
202
203
204
def patch(
    self,
    operations: list[dict[str, Any]],
    *,
    subresource: Literal["status"] | None = None,
    dry_run: bool = False,
) -> Self:
    """Apply a JSON Patch; cloudcoil.patches.diff adds UID/version preconditions."""
    config = context.active_config
    return config.client_for(self.__class__, sync=True).patch(
        self, operations, subresource=subresource, dry_run=dry_run
    )

save(dry_run=False)

Create or replace, preserving optimistic concurrency on existing resources.

Source code in cloudcoil/resources.py
229
230
231
232
233
234
235
236
237
238
239
240
241
242
def save(self, dry_run: bool = False) -> Self:
    """Create or replace, preserving optimistic concurrency on existing resources."""
    if self.name is None:
        return self.create(dry_run=dry_run)
    if self.resource_version is not None:
        return self.update(dry_run=dry_run)
    try:
        existing = self.fetch()
    except ResourceNotFound:
        return self.create(dry_run=dry_run)
    body = self.model_copy(deep=True)
    assert body.metadata is not None
    body.metadata.resource_version = existing.resource_version
    return body.update(dry_run=dry_run)

scale(replicas)

Scale the resource to the specified number of replicas.

Parameters:

Name Type Description Default
replicas int

The desired number of replicas

required

Returns:

Type Description
Self

The updated resource after scaling

Raises:

Type Description
ValueError

If the resource does not support scaling or metadata is not set

ResourceNotFound

If the resource does not exist

APIError

If the API request fails

Source code in cloudcoil/resources.py
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
def scale(self, replicas: int) -> Self:
    """Scale the resource to the specified number of replicas.

    Args:
        replicas: The desired number of replicas

    Returns:
        The updated resource after scaling

    Raises:
        ValueError: If the resource does not support scaling or metadata is not set
        ResourceNotFound: If the resource does not exist
        APIError: If the API request fails
    """
    config = context.active_config
    return config.client_for(self.__class__, sync=True).scale(
        self,
        replicas=replicas,
    )

ResourceList

Bases: BaseResource, Generic[T]

Source code in cloudcoil/resources.py
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
class ResourceList(BaseResource, Generic[T]):
    metadata: ListMeta | None = None
    items: list[T] = []
    _next_page_params: dict[str, Any] = {}
    _page_client: Any = None

    def __deepcopy__(self, memo=None) -> Self:
        memo = {} if memo is None else memo
        if self._page_client is not None:
            # Copy page data, but retain the live transport that owns pagination.
            memo[id(self._page_client)] = self._page_client
        return super().__deepcopy__(memo)

    @property
    def resource_class(self) -> type[T]:
        return self.__pydantic_generic_metadata__["args"][0]

    @model_validator(mode="after")
    def _validate_gvk(self):
        assert issubclass(self.resource_class, Resource)
        if self.api_version != self.resource_class.gvk().api_version:
            raise ValueError(f"api_version must be {self.resource_class.gvk().api_version}")
        if self.kind != self.resource_class.gvk().kind + "List":
            raise ValueError(f"kind must be {self.resource_class.gvk().kind + 'List'}")
        return self

    def has_next_page(self) -> bool:
        return bool(self.metadata and self.metadata.continue_)

    def get_next_page(self) -> "ResourceList[T]":
        if not self.has_next_page():
            raise ValueError("This resource list has no next page")
        if self._page_client is not None:
            from cloudcoil.client._api_client import APIClient

            if not isinstance(self._page_client, APIClient):
                raise TypeError("Use async_get_next_page() for an asynchronous resource list")
            return self._page_client.list(**self._next_page_params)
        config = context.active_config
        return config.client_for(self.resource_class, sync=True).list(**self._next_page_params)

    async def async_get_next_page(self) -> "ResourceList[T]":
        if not self.has_next_page():
            raise ValueError("This resource list has no next page")
        if self._page_client is not None:
            from cloudcoil.client._api_client import AsyncAPIClient

            if not isinstance(self._page_client, AsyncAPIClient):
                raise TypeError("Use get_next_page() for a synchronous resource list")
            return await self._page_client.list(**self._next_page_params)
        config = context.active_config
        return await (await config.async_client_for(self.resource_class)).list(
            **self._next_page_params
        )

    def __iter__(self):
        resource_list = self
        while True:
            for item in resource_list.items:
                yield item
            if not resource_list.has_next_page():
                break
            resource_list = resource_list.get_next_page()

    async def __aiter__(self):
        resource_list = self
        while True:
            for item in resource_list.items:
                yield item
            if not resource_list.has_next_page():
                break
            resource_list = await resource_list.async_get_next_page()

    def __len__(self):
        remaining = self.metadata.remaining_item_count if self.metadata else 0
        return len(self.items) + (remaining or 0)

Unstructured

Bases: Resource

Source code in cloudcoil/resources.py
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
class Unstructured(Resource):
    model_config = ConfigDict(extra="allow")

    def __getitem__(self, key: str) -> Any:
        for name, field in type(self).model_fields.items():
            if key in (name, field.alias):
                return getattr(self, name)
        if self.model_extra is not None and key in self.model_extra:
            return self.model_extra[key]
        raise KeyError(key)

    def __setitem__(self, key: str, value: Any) -> None:
        for name, field in type(self).model_fields.items():
            if key in (name, field.alias):
                setattr(self, name, value)
                return
        setattr(self, key, value)

    def __contains__(self, key: str) -> bool:
        return any(
            key in (name, field.alias) for name, field in type(self).model_fields.items()
        ) or (self.model_extra is not None and key in self.model_extra)

    @property
    def raw(self) -> dict:
        """Return a wire-format snapshot of the current validated model."""
        return self.model_dump(mode="json", by_alias=True, exclude_none=True)

raw property

Return a wire-format snapshot of the current validated model.

get_model(kind, *, api_version='')

Source code in cloudcoil/resources.py
765
766
def get_model(kind: str, *, api_version: str = "") -> Type[Resource]:
    return _Scheme.get(kind=kind, api_version=api_version)

parse(obj)

parse(obj: list[dict]) -> list[Resource]
parse(obj: dict) -> Resource
Source code in cloudcoil/resources.py
730
731
732
733
734
735
736
737
def parse(obj: dict | list[dict]) -> Resource | list[Resource]:
    if isinstance(obj, list):
        return [parse(o) for o in obj]
    resource = Resource.model_validate(obj)
    if not resource.api_version or not resource.kind:
        raise ValueError("Missing apiVersion or kind")
    kind = _Scheme.get(api_version=resource.api_version, kind=resource.kind)
    return kind.model_validate(obj)

parse_file(path, load_all=False)

parse_file(
    path: str | Path, load_all: Literal[True]
) -> list[Resource]
parse_file(
    path: str | Path, load_all: Literal[False] = False
) -> Resource
Source code in cloudcoil/resources.py
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
def parse_file(path: str | Path, load_all: bool = False) -> list[Resource] | Resource:
    content = Path(path).read_text()
    if not content.strip():
        raise ValueError("Empty YAML document")
    try:
        docs = [doc for doc in yaml.safe_load_all(content) if doc is not None]
        if not docs:
            raise ValueError("Empty YAML document")
        if not load_all:
            if len(docs) > 1:
                raise ValueError("Multiple YAML documents found when load_all=False")
            return parse(docs[0])
        return parse(docs)
    except yaml.YAMLError as e:
        raise ValueError(f"Failed to parse YAML: {e}")

Kubernetes clients; Config is loaded lazily to avoid the informer import cycle.

Config

Source code in cloudcoil/client/_config.py
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
class Config:
    def __init__(
        self,
        kubeconfig: Path | str | None = None,
        server: str | None = None,
        namespace: str | None = None,
        token: str | None = None,
        auth: Auth = None,
        cafile: Path | None = None,
        certfile: Path | None = None,
        keyfile: Path | None = None,
        context: str | None = None,
        skip_verify: bool | None = None,
        cache: Union[bool, "Cache"] = False,
    ) -> None:
        self._discovery_lock = threading.RLock()
        self._scope_lock = threading.RLock()
        self._async_scope_lock = asyncio.Lock()
        self._cache_users = 0
        self._cache_mode: str | None = None
        self._constructor_params: ConfigOptions = {
            "kubeconfig": kubeconfig,
            "server": server,
            "namespace": namespace,
            "token": token,
            "auth": auth,
            "cafile": cafile,
            "certfile": certfile,
            "keyfile": keyfile,
            "context": context,
            "skip_verify": skip_verify,
        }
        self.server = None
        self.namespace = "default"
        self.auth: Auth = None
        self.cafile = None
        self.certfile = None
        self.keyfile = None
        self.token = None
        self.skip_verify = False
        self.kubeconfig_path = None  # Store original kubeconfig path
        tempdir = self._tempdir = tempfile.TemporaryDirectory()
        kubeconfig = kubeconfig or os.environ.get("KUBECONFIG")
        if kubeconfig:
            kubeconfig = Path(kubeconfig)
            if not kubeconfig.is_file():
                logger.error("Kubeconfig not found: %s", kubeconfig)
                raise ValueError(f"Kubeconfig {kubeconfig} is not a file")
        else:
            kubeconfig = DEFAULT_KUBECONFIG
            logger.debug("Using default kubeconfig: %s", kubeconfig)

        if kubeconfig.is_file():
            self.kubeconfig_path = kubeconfig  # Store the kubeconfig path
            logger.debug("Loading kubeconfig from: %s", kubeconfig)
            kubeconfig_data = yaml.safe_load(kubeconfig.read_text())
            if "clusters" not in kubeconfig_data:
                logger.error("Invalid kubeconfig: missing clusters section")
                raise ValueError(f"Kubeconfig {kubeconfig} does not have clusters")
            if "contexts" not in kubeconfig_data:
                logger.error("Invalid kubeconfig: missing contexts section")
                raise ValueError(f"Kubeconfig {kubeconfig} does not have contexts")
            if "users" not in kubeconfig_data:
                logger.error("Invalid kubeconfig: missing users section")
                raise ValueError(f"Kubeconfig {kubeconfig} does not have users")
            if not context and "current-context" not in kubeconfig_data:
                logger.error("Invalid kubeconfig: no current-context specified")
                raise ValueError(f"Kubeconfig {kubeconfig} does not have current-context")
            current_context = context or kubeconfig_data["current-context"]
            self._constructor_params["context"] = current_context
            logger.debug("Using context: %s", current_context)

            for data in kubeconfig_data["contexts"]:
                if data["name"] == current_context:
                    break
            else:
                logger.error("Context not found in kubeconfig: %s", current_context)
                raise ValueError(f"Kubeconfig {kubeconfig} does not have context {current_context}")
            context_data = data["context"]

            for data in kubeconfig_data["clusters"]:
                if data["name"] == context_data["cluster"]:
                    break
            else:
                logger.error("Cluster not found in kubeconfig: %s", context_data["cluster"])
                raise ValueError(
                    f"Kubeconfig {kubeconfig} does not have cluster {context_data['cluster']}"
                )
            cluster_data = data["cluster"]

            for data in kubeconfig_data["users"]:
                if data["name"] == context_data["user"]:
                    break
            else:
                logger.error("User not found in kubeconfig: %s", context_data["user"])
                raise ValueError(
                    f"Kubeconfig {kubeconfig} does not have user {context_data['user']}"
                )
            user_data = data["user"]

            self.server = cluster_data["server"]
            logger.debug("Using server: %s", self.server)

            if "certificate-authority" in cluster_data:
                self.cafile = kubeconfig.parent / cluster_data["certificate-authority"]
                logger.debug("Using CA file: %s", self.cafile)
            if "certificate-authority-data" in cluster_data:
                # Write certificate to disk at a temporary location and use it
                cafile = Path(tempdir.name) / "ca.crt"
                cafile.write_bytes(base64.b64decode(cluster_data["certificate-authority-data"]))
                self.cafile = cafile
                logger.debug("Using temporary CA file: %s", self.cafile)

            if "insecure-skip-tls-verify" in cluster_data:
                self.skip_verify = cluster_data["insecure-skip-tls-verify"]
                if self.skip_verify:
                    logger.warning("TLS verification disabled")

            if "namespace" in context_data:
                self.namespace = context_data["namespace"]
                logger.debug("Using namespace from context: %s", self.namespace)

            if "exec" in user_data:
                logger.debug("Using exec auth provider")
                self.auth = ExecAuthenticator(user_data["exec"])
            elif "token" in user_data:
                logger.debug("Using static token auth")
                self.token = user_data["token"]
            elif "client-certificate" in user_data and "client-key" in user_data:
                self.certfile = kubeconfig.parent / user_data["client-certificate"]
                self.keyfile = kubeconfig.parent / user_data["client-key"]
                logger.debug(
                    "Using client certificate auth: cert=%s key=%s", self.certfile, self.keyfile
                )
            elif "client-certificate-data" in user_data and "client-key-data" in user_data:
                # Write client certificate and key to disk at a temporary location
                # and use them
                client_cert = Path(tempdir.name) / "client.crt"
                client_cert.write_bytes(base64.b64decode(user_data["client-certificate-data"]))
                client_key = Path(tempdir.name) / "client.key"
                client_key.write_bytes(base64.b64decode(user_data["client-key-data"]))
                self.certfile = client_cert
                self.keyfile = client_key
                logger.debug(
                    "Using temporary client certificate auth: cert=%s key=%s",
                    self.certfile,
                    self.keyfile,
                )

        elif INCLUSTER_TOKEN_PATH.is_file():
            logger.debug("Detected in-cluster environment")
            self.server = "https://kubernetes.default.svc"
            self.namespace = INCLUSTER_NAMESPACE_PATH.read_text()
            self.token = INCLUSTER_TOKEN_PATH.read_text()
            if INCLUSTER_CERT_PATH.is_file():
                self.cafile = INCLUSTER_CERT_PATH
                logger.debug("Using in-cluster CA file: %s", self.cafile)

        self.server = server or self.server or "https://localhost:6443"
        self.namespace = namespace or self.namespace
        self.token = token or self.token
        self.auth = auth or self.auth
        self.cafile = cafile or self.cafile
        self.certfile = certfile or self.certfile
        self.keyfile = keyfile or self.keyfile
        self.skip_verify = self.skip_verify if skip_verify is None else skip_verify

        ctx: ssl.SSLContext | None = None
        if self.cafile:
            logger.debug("Creating SSL context with CA file: %s", self.cafile)
            ctx = ssl.create_default_context(cafile=self.cafile)
        else:
            logger.debug("Using default SSL context")
            ctx = _new_ssl_context()

        if self.certfile:
            logger.debug("Loading client certificate: cert=%s key=%s", self.certfile, self.keyfile)
            ctx.load_cert_chain(certfile=self.certfile, keyfile=self.keyfile)

        if self.skip_verify:
            logger.warning("Using insecure SSL context (verify=False)")
            ctx.check_hostname = False
            ctx.verify_mode = ssl.CERT_NONE

        headers = {
            "User-Agent": f"cloudcoil/{__version__} ({platform.platform()}) python/{platform.python_version()}",
        }
        if self.token:
            logger.debug("Adding token authentication to headers")
            headers["Authorization"] = f"Bearer {self.token}"

        logger.debug("Creating HTTP clients for server %s", self.server)
        self.client = httpx.Client(
            verify=ctx, auth=self.auth or None, base_url=self.server, headers=headers
        )
        self.async_client = httpx.AsyncClient(
            verify=ctx, auth=self.auth or None, base_url=self.server, headers=headers
        )
        self._rest_mapping: dict[GVK, Any] = {}

        # Parse cache parameter
        if cache is True:
            self.cache = Cache()  # Use defaults
        elif cache is False:
            self.cache = Cache(enabled=False)
        elif isinstance(cache, Cache):
            self.cache = cache
        else:
            raise ValueError("cache must be bool or Cache object")

        # Set client factory on cache if enabled
        # Pass the client_for method as the factory
        if self.cache.enabled:
            self.cache.set_client_factory(self.client_for)

    def _create_rest_mapper(self):
        logger.debug("Fetching Kubernetes version")
        version_response = self.client.get("/version")
        if version_response.status_code != 200:
            logger.error("Failed to get Kubernetes version: %s", version_response.text)
            raise ValueError(f"Failed to get version: {version_response.text}")
        version_data = version_response.json()
        major = int("".join(filter(str.isdigit, version_data["major"])))
        minor = int("".join(filter(str.isdigit, version_data["minor"])))
        logger.debug("Connected to Kubernetes %s.%s", major, minor)

        if major > 1 or (major == 1 and minor >= 30):
            try:
                logger.debug("Attempting aggregated API discovery")
                if self._try_aggregated_discovery():
                    return
            except Exception as e:
                logger.debug("Aggregated discovery failed, falling back to traditional: %s", e)

        logger.debug("Using traditional API discovery")
        self._traditional_discovery()

    def _try_aggregated_discovery(self) -> bool:
        logger.debug("Fetching core API groups")
        api_response = self.client.get(
            "/api",
            headers={
                "Accept": "application/json;v=v2;g=apidiscovery.k8s.io;as=APIGroupDiscoveryList"
            },
        )
        if api_response.status_code != 200:
            logger.debug("Core API groups not available in aggregated format")
            return False
        self._process_api_discovery(api_response.json())

        logger.debug("Fetching extension API groups")
        apis_response = self.client.get(
            "/apis",
            headers={
                "Accept": "application/json;v=v2;g=apidiscovery.k8s.io;as=APIGroupDiscoveryList"
            },
        )
        if apis_response.status_code != 200:
            logger.debug("Extension API groups not available in aggregated format")
            return False
        self._process_api_discovery(apis_response.json())
        return True

    def _traditional_discovery(self):
        logger.debug("Fetching core API versions")
        api_response = self.client.get("/api")
        if api_response.status_code == 200:
            for version in api_response.json().get("versions", []):
                logger.debug("Fetching resources for core API version: %s", version)
                version_response = self.client.get(f"/api/{version}")
                if version_response.status_code == 200:
                    self._process_api_resources("", version, version_response.json())

        logger.debug("Fetching API groups")
        apis_response = self.client.get("/apis")
        if apis_response.status_code != 200:
            logger.error("Failed to get API groups: %s", apis_response.text)
            raise ValueError(f"Failed to get APIs: {apis_response.text}")

        for group in apis_response.json().get("groups", []):
            group_name = group["name"]
            for version_data in group["versions"]:
                version = version_data["version"]
                logger.debug("Fetching resources for API group: %s/%s", group_name, version)
                group_response = self.client.get(f"/apis/{group_name}/{version}")
                if group_response.status_code == 200:
                    self._process_api_resources(group_name, version, group_response.json())

    def _process_api_discovery(self, api_discovery):
        if not isinstance(api_discovery, dict) or "items" not in api_discovery:
            logger.debug("Invalid API discovery response format")
            return

        for api in api_discovery["items"]:
            group = api.get("metadata", {}).get("name", "")
            for version_data in api.get("versions", []):
                version = version_data.get("version")
                if not version:
                    continue

                for resource_data in version_data.get("resources", []):
                    kind = resource_data.get("responseKind", {}).get("kind")
                    resource = resource_data.get("resource")
                    scope = resource_data.get("scope")
                    subresources = list(
                        map(lambda sr: sr["subresource"], resource_data.get("subresources", {}))
                    )

                    if not all([kind, resource, scope]):
                        continue

                    namespaced = scope == "Namespaced"
                    api_version = f"{group}/{version}" if group != "" else version
                    logger.debug(
                        "Registered API resource: %s/%s (namespaced=%s)",
                        api_version,
                        kind,
                        namespaced,
                    )
                    self._rest_mapping[GVK(api_version=api_version, kind=kind)] = {
                        "namespaced": namespaced,
                        "resource": resource,
                        "subresources": subresources,
                    }

    def _process_api_resources(self, group: str, version: str, data: dict):
        api_version = f"{group}/{version}" if group else version

        for resource in data.get("resources", []):
            if "/" in resource["name"]:
                continue

            kind = resource["kind"]
            namespaced = resource["namespaced"]
            resource_name = resource["name"]
            subresources: list[str] = []

            logger.debug(
                "Registered API resource: %s/%s (namespaced=%s)",
                api_version,
                kind,
                namespaced,
            )
            self._rest_mapping[GVK(api_version=api_version, kind=kind)] = {
                "namespaced": namespaced,
                "resource": resource_name,
                "subresources": subresources,
            }
            for subresource in data.get("resources", []):
                if subresource["name"].startswith(f"{resource_name}/"):
                    subresources.append(subresource["name"].split("/", 1)[1])

    @overload
    def client_for(
        self, resource: Type[T], sync: Literal[True] = True, cached: bool | None = None
    ) -> APIClient[T]: ...

    @overload
    def client_for(
        self, resource: Type[T], sync: Literal[False] = False, cached: bool | None = None
    ) -> AsyncAPIClient[T]: ...

    def client_for(
        self, resource: Type[T], sync: Literal[False, True] = True, cached: bool | None = None
    ) -> APIClient[T] | AsyncAPIClient[T]:
        """Get a client for the specified resource type.

        Args:
            resource: The resource type to get a client for
            sync: Whether to return a sync or async client
            cached: Whether to use caching. If None, uses cache config setting.
                   If True, forces cached client (requires cache to be enabled).
                   If False, forces non-cached client.

        Returns:
            A client that may use caching based on configuration
        """
        self.initialize()
        if not issubclass(resource, Resource):
            logger.error("Invalid resource type: %s", resource)
            raise ValueError(f"Resource {resource} is not a cloudcoil.Resource")
        gvk = resource.gvk()
        if gvk not in self._rest_mapping:
            logger.error("Resource not registered with API server: %s", gvk)
            raise ValueError(f"Resource with {gvk=} is not registered with the server")

        logger.debug("Creating %s client for %s", "sync" if sync else "async", gvk)

        # Determine if we should use caching
        use_cache = cached if cached is not None else self.cache.enabled

        # If caching requested, check if we can provide it
        if use_cache:
            if not self.cache.enabled:
                raise ValueError("Cannot create cached client when cache is disabled")

            informer = self.cache.get_informer(resource, sync=sync)
            if informer:
                strict = self.cache.mode == "strict"
                if sync:
                    return CachedClient(
                        api_version=gvk.api_version,
                        kind=resource,
                        resource=self._rest_mapping[gvk]["resource"],
                        namespaced=self._rest_mapping[gvk]["namespaced"],
                        subresources=self._rest_mapping[gvk]["subresources"],
                        default_namespace=self.namespace,
                        client=self.client,
                        informer=informer,  # type: ignore[arg-type]
                        strict=strict,
                    )
                else:
                    return AsyncCachedClient(
                        api_version=gvk.api_version,
                        kind=resource,
                        resource=self._rest_mapping[gvk]["resource"],
                        namespaced=self._rest_mapping[gvk]["namespaced"],
                        subresources=self._rest_mapping[gvk]["subresources"],
                        default_namespace=self.namespace,
                        client=self.async_client,
                        informer=informer,  # type: ignore[arg-type]
                        strict=strict,
                    )

        # Return non-cached client
        if sync:
            return APIClient(
                api_version=gvk.api_version,
                kind=resource,
                resource=self._rest_mapping[gvk]["resource"],
                namespaced=self._rest_mapping[gvk]["namespaced"],
                subresources=self._rest_mapping[gvk]["subresources"],
                default_namespace=self.namespace,
                client=self.client,
            )
        return AsyncAPIClient(
            api_version=gvk.api_version,
            kind=resource,
            resource=self._rest_mapping[gvk]["resource"],
            namespaced=self._rest_mapping[gvk]["namespaced"],
            subresources=self._rest_mapping[gvk]["subresources"],
            default_namespace=self.namespace,
            client=self.async_client,
        )

    def set_default(self) -> None:
        logger.debug("Setting as default config")
        context.set_default(self)

    def initialize(self):
        with self._discovery_lock:
            if not self._rest_mapping:
                logger.debug("Initializing API resource mapping")
                try:
                    self._create_rest_mapper()
                except BaseException:
                    # A failed discovery must be retried, not mistaken for a
                    # completed mapping because it yielded some resources.
                    self._rest_mapping.clear()
                    raise

    async def async_initialize(self) -> None:
        """Discover resources without blocking the event loop."""
        # Discovery populates the map incrementally. A nonempty map does not
        # prove another thread has finished, so acquire its lock off-loop.
        await asyncio.to_thread(self.initialize)

    async def async_client_for(
        self, resource: Type[T], cached: bool | None = None
    ) -> AsyncAPIClient[T]:
        """Get an async client without blocking on discovery or concurrent refresh."""

        def create() -> AsyncAPIClient[T]:
            with self._discovery_lock:
                return self.client_for(resource, sync=False, cached=cached)

        return await asyncio.to_thread(create)

    def activate(self):
        logger.debug("Activating config")
        self.__enter__()

    def deactivate(self):
        logger.debug("Deactivating config")
        self.__exit__()

    def __enter__(self):
        self.initialize()
        context._enter(self)
        try:
            if self.cache.enabled:
                with self._scope_lock:
                    if self._cache_mode == "async":
                        raise RuntimeError("Cannot mix sync and async scopes on a cached Config")
                    if self._cache_users == 0:
                        self._cache_mode = "sync"
                        try:
                            self.cache.start()
                            if self.cache.wait_for_sync and not self.cache.wait():
                                if self.cache.mode == "strict":
                                    raise APIError(
                                        f"Cache failed to sync within {self.cache.sync_timeout}s"
                                    )
                                logger.warning("Cache sync timeout, continuing with fallback")
                        except BaseException:
                            self._cache_mode = None
                            self.cache.stop()
                            raise
                    self._cache_mode = "sync"
                    self._cache_users += 1
        except BaseException:
            context._exit(self)
            raise
        return self

    def __exit__(self, *_):
        configs = context.configs
        if not configs or configs[-1] is not self:
            raise RuntimeError("Configurations must be deactivated in reverse activation order")
        try:
            if self.cache.enabled:
                with self._scope_lock:
                    self._cache_users -= 1
                    if self._cache_users == 0:
                        self._cache_mode = None
                        self.cache.stop()
        finally:
            context._exit(self)

    async def __aenter__(self):
        await self.async_initialize()
        context._enter(self)
        try:
            if self.cache.enabled:
                async with self._async_scope_lock:
                    if self._cache_mode == "sync":
                        raise RuntimeError("Cannot mix sync and async scopes on a cached Config")
                    if self._cache_users == 0:
                        self._cache_mode = "async"
                        try:
                            await self.cache.async_start()
                            if self.cache.wait_for_sync and not await self.cache.async_wait():
                                if self.cache.mode == "strict":
                                    raise APIError(
                                        f"Cache failed to sync within {self.cache.sync_timeout}s"
                                    )
                                logger.warning("Cache sync timeout, continuing with fallback")
                        except BaseException:
                            self._cache_mode = None
                            await self.cache.async_stop()
                            raise
                    self._cache_mode = "async"
                    self._cache_users += 1
        except BaseException:
            context._exit(self)
            raise
        return self

    async def __aexit__(self, *_):
        configs = context.configs
        if not configs or configs[-1] is not self:
            raise RuntimeError("Configurations must be deactivated in reverse activation order")
        try:
            if self.cache.enabled:
                async with self._async_scope_lock:
                    self._cache_users -= 1
                    if self._cache_users == 0:
                        self._cache_mode = None
                        await self.cache.async_stop()
        finally:
            context._exit(self)

    def refresh_api_resources(self) -> None:
        logger.debug("Refreshing API resource mapping")
        with self._discovery_lock:
            self._rest_mapping.clear()
            self.initialize()

    def clone(self, **overrides: Unpack[ConfigOptions]) -> "Config":
        """Create a new Config instance with the same parameters but with specified overrides.

        This method creates a new Config using the original kubeconfig path if available,
        ensuring proper certificate handling, and applies any specified overrides.

        Args:
            **overrides: Any Config constructor parameters to override

        Returns:
            A new Config instance with the same base configuration but with overrides applied
        """
        base_params: ConfigOptions = self._constructor_params.copy()
        base_params.update(
            {
                "kubeconfig": self.kubeconfig_path,
                "server": self.server,
                "namespace": self.namespace,
                "token": self.token,
                "auth": self.auth,
                "skip_verify": self.skip_verify,
                "cache": Cache(**self.cache.model_dump()),
            }
        )
        base_params.update(overrides)
        if "context" in overrides or "kubeconfig" in overrides:
            for key in ("server", "namespace", "token", "auth", "skip_verify"):
                if key not in overrides:
                    base_params[key] = self._constructor_params[key]
            if "kubeconfig" in overrides and "context" not in overrides:
                base_params["context"] = None

        return Config(**base_params)

    def with_cache(self, cache: Union[bool, "Cache"]) -> "Config":
        """Create a new Config instance with the same parameters but different cache settings."""
        return self.clone(cache=cache)

async_client_for(resource, cached=None) async

Get an async client without blocking on discovery or concurrent refresh.

Source code in cloudcoil/client/_config.py
604
605
606
607
608
609
610
611
612
613
async def async_client_for(
    self, resource: Type[T], cached: bool | None = None
) -> AsyncAPIClient[T]:
    """Get an async client without blocking on discovery or concurrent refresh."""

    def create() -> AsyncAPIClient[T]:
        with self._discovery_lock:
            return self.client_for(resource, sync=False, cached=cached)

    return await asyncio.to_thread(create)

async_initialize() async

Discover resources without blocking the event loop.

Source code in cloudcoil/client/_config.py
598
599
600
601
602
async def async_initialize(self) -> None:
    """Discover resources without blocking the event loop."""
    # Discovery populates the map incrementally. A nonempty map does not
    # prove another thread has finished, so acquire its lock off-loop.
    await asyncio.to_thread(self.initialize)

client_for(resource, sync=True, cached=None)

client_for(
    resource: Type[T],
    sync: Literal[True] = True,
    cached: bool | None = None,
) -> APIClient[T]
client_for(
    resource: Type[T],
    sync: Literal[False] = False,
    cached: bool | None = None,
) -> AsyncAPIClient[T]

Get a client for the specified resource type.

Parameters:

Name Type Description Default
resource Type[T]

The resource type to get a client for

required
sync Literal[False, True]

Whether to return a sync or async client

True
cached bool | None

Whether to use caching. If None, uses cache config setting. If True, forces cached client (requires cache to be enabled). If False, forces non-cached client.

None

Returns:

Type Description
APIClient[T] | AsyncAPIClient[T]

A client that may use caching based on configuration

Source code in cloudcoil/client/_config.py
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
def client_for(
    self, resource: Type[T], sync: Literal[False, True] = True, cached: bool | None = None
) -> APIClient[T] | AsyncAPIClient[T]:
    """Get a client for the specified resource type.

    Args:
        resource: The resource type to get a client for
        sync: Whether to return a sync or async client
        cached: Whether to use caching. If None, uses cache config setting.
               If True, forces cached client (requires cache to be enabled).
               If False, forces non-cached client.

    Returns:
        A client that may use caching based on configuration
    """
    self.initialize()
    if not issubclass(resource, Resource):
        logger.error("Invalid resource type: %s", resource)
        raise ValueError(f"Resource {resource} is not a cloudcoil.Resource")
    gvk = resource.gvk()
    if gvk not in self._rest_mapping:
        logger.error("Resource not registered with API server: %s", gvk)
        raise ValueError(f"Resource with {gvk=} is not registered with the server")

    logger.debug("Creating %s client for %s", "sync" if sync else "async", gvk)

    # Determine if we should use caching
    use_cache = cached if cached is not None else self.cache.enabled

    # If caching requested, check if we can provide it
    if use_cache:
        if not self.cache.enabled:
            raise ValueError("Cannot create cached client when cache is disabled")

        informer = self.cache.get_informer(resource, sync=sync)
        if informer:
            strict = self.cache.mode == "strict"
            if sync:
                return CachedClient(
                    api_version=gvk.api_version,
                    kind=resource,
                    resource=self._rest_mapping[gvk]["resource"],
                    namespaced=self._rest_mapping[gvk]["namespaced"],
                    subresources=self._rest_mapping[gvk]["subresources"],
                    default_namespace=self.namespace,
                    client=self.client,
                    informer=informer,  # type: ignore[arg-type]
                    strict=strict,
                )
            else:
                return AsyncCachedClient(
                    api_version=gvk.api_version,
                    kind=resource,
                    resource=self._rest_mapping[gvk]["resource"],
                    namespaced=self._rest_mapping[gvk]["namespaced"],
                    subresources=self._rest_mapping[gvk]["subresources"],
                    default_namespace=self.namespace,
                    client=self.async_client,
                    informer=informer,  # type: ignore[arg-type]
                    strict=strict,
                )

    # Return non-cached client
    if sync:
        return APIClient(
            api_version=gvk.api_version,
            kind=resource,
            resource=self._rest_mapping[gvk]["resource"],
            namespaced=self._rest_mapping[gvk]["namespaced"],
            subresources=self._rest_mapping[gvk]["subresources"],
            default_namespace=self.namespace,
            client=self.client,
        )
    return AsyncAPIClient(
        api_version=gvk.api_version,
        kind=resource,
        resource=self._rest_mapping[gvk]["resource"],
        namespaced=self._rest_mapping[gvk]["namespaced"],
        subresources=self._rest_mapping[gvk]["subresources"],
        default_namespace=self.namespace,
        client=self.async_client,
    )

clone(**overrides)

Create a new Config instance with the same parameters but with specified overrides.

This method creates a new Config using the original kubeconfig path if available, ensuring proper certificate handling, and applies any specified overrides.

Parameters:

Name Type Description Default
**overrides Unpack[ConfigOptions]

Any Config constructor parameters to override

{}

Returns:

Type Description
Config

A new Config instance with the same base configuration but with overrides applied

Source code in cloudcoil/client/_config.py
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
def clone(self, **overrides: Unpack[ConfigOptions]) -> "Config":
    """Create a new Config instance with the same parameters but with specified overrides.

    This method creates a new Config using the original kubeconfig path if available,
    ensuring proper certificate handling, and applies any specified overrides.

    Args:
        **overrides: Any Config constructor parameters to override

    Returns:
        A new Config instance with the same base configuration but with overrides applied
    """
    base_params: ConfigOptions = self._constructor_params.copy()
    base_params.update(
        {
            "kubeconfig": self.kubeconfig_path,
            "server": self.server,
            "namespace": self.namespace,
            "token": self.token,
            "auth": self.auth,
            "skip_verify": self.skip_verify,
            "cache": Cache(**self.cache.model_dump()),
        }
    )
    base_params.update(overrides)
    if "context" in overrides or "kubeconfig" in overrides:
        for key in ("server", "namespace", "token", "auth", "skip_verify"):
            if key not in overrides:
                base_params[key] = self._constructor_params[key]
        if "kubeconfig" in overrides and "context" not in overrides:
            base_params["context"] = None

    return Config(**base_params)

with_cache(cache)

Create a new Config instance with the same parameters but different cache settings.

Source code in cloudcoil/client/_config.py
749
750
751
def with_cache(self, cache: Union[bool, "Cache"]) -> "Config":
    """Create a new Config instance with the same parameters but different cache settings."""
    return self.clone(cache=cache)

APIClient

Bases: _BaseAPIClient[T]

Source code in cloudcoil/client/_api_client.py
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
class APIClient(_BaseAPIClient[T]):
    def __init__(
        self,
        api_version: str,
        kind: Type[T],
        resource: str,
        subresources: list[str],
        default_namespace: str,
        namespaced: bool,
        client: httpx.Client,
    ) -> None:
        super().__init__(api_version, kind, resource, subresources, default_namespace, namespaced)
        self._client = client

    def get(self, name: str, namespace: str | None = None) -> T:
        namespace = namespace or self.default_namespace
        url = self._build_url(name=name, namespace=namespace)
        response = self._client.get(url)
        return self._handle_get_response(response, namespace, name)

    def create(self, body: T, dry_run: bool = False) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata must be set for {body=}")
        namespace = body.namespace or self.default_namespace
        url = self._build_url(namespace=namespace)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        response = self._client.post(
            url, json=body.model_dump(mode="json", by_alias=True, exclude_none=True), params=params
        )
        return self._handle_create_response(response)

    def update(self, body: T, dry_run: bool = False) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata must be set for {body=}")
        namespace = body.namespace or self.default_namespace
        name = body.name
        if not name:
            raise ValueError("metadata.name must be set for update operations")
        url = self._build_url(namespace=namespace, name=name)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        response = self._client.put(
            url, json=body.model_dump(mode="json", by_alias=True, exclude_none=True), params=params
        )
        return self._handle_create_response(response)

    def patch(
        self,
        body: T,
        operations: list[dict[str, Any]],
        *,
        subresource: Literal["status"] | None = None,
        dry_run: bool = False,
    ) -> T:
        """Apply an RFC 6902 JSON Patch; include tests for optimistic concurrency."""
        url = self._patch_url(body, subresource)
        response = self._client.patch(
            url,
            json=operations,
            headers={"Content-Type": "application/json-patch+json"},
            params={"dryRun": "All"} if dry_run else {},
        )
        return self._handle_create_response(response)

    def update_status(self, body: T, dry_run: bool = False) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata must be set for {body=}")
        if "status" not in self.subresources:
            raise ValueError(f"Resource {body.gvk().kind} does not support status updates")
        namespace = body.namespace or self.default_namespace
        name = body.name
        if not name:
            raise ValueError("metadata.name must be set for update operations")
        url = f"{self._build_url(namespace=namespace, name=name)}/status"
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        response = self._client.put(
            url, json=body.model_dump(mode="json", by_alias=True, exclude_none=True), params=params
        )
        return self._handle_create_response(response)

    def delete(
        self,
        name: str,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
        *,
        uid: str | None = None,
        resource_version: str | None = None,
    ) -> T | Status:
        namespace = namespace or self.default_namespace
        url = self._build_url(name=name, namespace=namespace)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        if propagation_policy:
            params["propagationPolicy"] = propagation_policy.capitalize()
        if grace_period_seconds is not None:
            params["gracePeriodSeconds"] = grace_period_seconds
        preconditions = {
            key: value
            for key, value in {
                "uid": uid,
                "resourceVersion": resource_version,
            }.items()
            if value is not None
        }
        if preconditions:
            response = self._client.request(
                "DELETE",
                url,
                params=params,
                json={"apiVersion": "v1", "kind": "DeleteOptions", "preconditions": preconditions},
            )
        else:
            response = self._client.delete(url, params=params)
        return self._handle_delete_response(response, namespace, name)

    def remove(
        self,
        body: T,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
    ) -> T | Status:
        if not (body.metadata and body.metadata.name):
            raise ValueError(f"metadata.name must be set for {body=}")
        namespace = body.metadata.namespace or self.default_namespace
        name = body.metadata.name
        return self.delete(
            name,
            namespace,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
        )

    def list(
        self,
        namespace: str | None = None,
        all_namespaces: bool = False,
        continue_: None | str = None,
        field_selector: str | None = None,
        label_selector: str | None = None,
        limit: int = DEFAULT_PAGE_LIMIT,
    ) -> ResourceList[T]:
        namespace = namespace or self.default_namespace
        if all_namespaces:
            namespace = None
        url = self._build_url(namespace=namespace)
        params: dict[str, str | int] = {}
        if continue_:
            params["continue"] = continue_
        if field_selector:
            params["fieldSelector"] = field_selector
        if label_selector:
            params["labelSelector"] = label_selector
        if limit:
            params["limit"] = limit
        response = self._client.get(url, params=params)
        self._raise_for_status(response)
        output = ResourceList[self.kind].model_validate_json(response.content)  # type: ignore
        assert output.metadata
        output._page_client = self
        output._next_page_params = {
            "namespace": namespace,
            "all_namespaces": all_namespaces,
            "continue_": output.metadata.continue_,
            "field_selector": field_selector,
            "label_selector": label_selector,
            "limit": limit,
        }
        return output

    def delete_all(
        self,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
        label_selector: str | None = None,
        field_selector: str | None = None,
    ) -> ResourceList[T]:
        namespace = namespace or self.default_namespace
        url = self._build_url(namespace=namespace)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        if propagation_policy:
            params["propagationPolicy"] = propagation_policy.capitalize()
        if grace_period_seconds is not None:
            params["gracePeriodSeconds"] = grace_period_seconds
        if label_selector:
            params["labelSelector"] = label_selector
        if field_selector:
            params["fieldSelector"] = field_selector
        logger.warning(
            "Deleting all resources: kind=%s namespace=%s policy=%s grace=%s label_selector=%s field_selector=%s",
            self.kind.gvk().kind,
            namespace,
            propagation_policy,
            grace_period_seconds,
            label_selector,
            field_selector,
        )
        response = self._client.delete(url, params=params)
        self._raise_for_status(response)
        return ResourceList[self.kind].model_validate_json(response.content)  # type: ignore

    def watch(
        self,
        namespace: str | None = None,
        all_namespaces: bool = False,
        field_selector: str | None = None,
        label_selector: str | None = None,
        resource_version: str | None = None,
        timeout_seconds: int | None = None,
        _stop_event: threading.Event | None = None,
        _raise_on_expired: bool = False,
    ) -> Generator[tuple[WatchEvent, T] | tuple[BookmarkEvent, Unstructured], None, None]:
        retry_count = 0
        curr_resource_version = resource_version
        kind_name = self.kind.gvk().kind

        while _stop_event is None or not _stop_event.is_set():
            try:
                url, params = self._build_watch_params(
                    namespace=namespace,
                    all_namespaces=all_namespaces,
                    field_selector=field_selector,
                    label_selector=label_selector,
                    resource_version=curr_resource_version,
                    timeout=timeout_seconds or _WATCH_TIMEOUT_SECONDS,
                )

                with self._client.stream(
                    "GET",
                    url,
                    params=params,
                    timeout=(timeout_seconds or _WATCH_TIMEOUT_SECONDS) + 5,
                ) as response:
                    if response.status_code == 410:  # Gone
                        if _raise_on_expired:
                            raise WatchExpired(
                                "Watch history expired; relist required", status_code=410
                            )
                        logger.debug(
                            "Watch resource version expired, restarting: kind=%s namespace=%s",
                            kind_name,
                            namespace,
                        )
                        curr_resource_version = None
                        continue

                    response.raise_for_status()
                    retry_count = 0  # Reset retry counter on successful connection

                    for line in response.iter_lines():
                        if not line:
                            continue
                        event = json.loads(line)
                        type_ = event["type"]

                        if type_ == "ERROR":
                            if _raise_on_expired and (
                                event["object"].get("code") == 410
                                or event["object"].get("reason") in {"Expired", "Gone"}
                            ):
                                raise WatchExpired(
                                    "Watch history expired; relist required", status_code=410
                                )
                            code = event["object"].get("code")
                            if (
                                isinstance(code, int)
                                and 400 <= code < 500
                                and code not in {408, 410, 429}
                            ):
                                raise APIError(event["object"], status_code=code)
                            if "status" in event["object"]:
                                status = event["object"]["status"]
                                if status == "Failure":
                                    reason = event["object"].get("reason", "")
                                    if reason == "Expired":
                                        logger.debug(
                                            "Watch resource version expired, restarting: kind=%s namespace=%s",
                                            kind_name,
                                            namespace,
                                        )
                                        curr_resource_version = None
                                        break
                            logger.error("Watch error event received: %s", event)
                            raise WatchError(f"Watch error: {event}")

                        # Handle bookmark events specially - they only contain minimal data
                        if type_ == "BOOKMARK":
                            # For bookmark events, create an Unstructured object to avoid validation errors
                            bookmark_obj: Unstructured = Unstructured.model_validate(
                                event["object"]
                            )
                            if bookmark_obj.metadata and bookmark_obj.metadata.resource_version:
                                curr_resource_version = bookmark_obj.metadata.resource_version
                            yield type_, bookmark_obj
                        else:
                            obj = self.kind.model_validate(event["object"])
                            if obj.metadata and obj.metadata.resource_version:
                                curr_resource_version = obj.metadata.resource_version
                            yield type_, obj

            except WatchExpired:
                raise
            except (httpx.RequestError, httpx.HTTPStatusError, WatchError) as e:
                if isinstance(e, httpx.HTTPStatusError):
                    if e.response.status_code == 410:  # Gone
                        logger.debug(
                            "Watch resource version expired, restarting: kind=%s namespace=%s",
                            self.kind.gvk().kind,
                            namespace,
                        )
                        curr_resource_version = None
                        continue
                    if e.response.status_code == 404:  # Not Found
                        logger.error(
                            "Watch endpoint not found: kind=%s namespace=%s",
                            self.kind.gvk().kind,
                            namespace,
                        )
                        raise ResourceNotFound("Watch endpoint not found", status_code=404) from e
                    if 400 <= e.response.status_code < 500 and e.response.status_code not in {
                        408,
                        429,
                    }:
                        raise APIError(str(e), status_code=e.response.status_code) from e

                retry_count += 1
                backoff = self._get_backoff_time(retry_count)
                logger.warning(
                    "Watch connection failed, retrying in %0.1fs: %s",
                    backoff,
                    str(e),
                )
                if _stop_event is not None:
                    _stop_event.wait(backoff)
                else:
                    time.sleep(backoff)

    def wait_for(
        self,
        resource: T,
        predicates: dict[str, WaitPredicate],
        timeout: float | None = None,
    ) -> str:
        if not resource.name:
            raise ValueError("metadata.name must be set to wait for a resource")
        if not predicates:
            raise ValueError("At least one wait predicate is required")
        if timeout is not None and timeout <= 0:
            raise WaitTimeout("Timeout waiting for condition")

        class WatchResult:
            def __init__(self) -> None:
                self.predicate_name: str | None = None
                self.error: Exception | None = None

        result = WatchResult()
        stop_event = threading.Event()

        def watch_and_evaluate() -> None:
            try:
                for event_type, obj in self.watch(
                    namespace=resource.namespace,
                    field_selector=f"metadata.name={resource.name}",
                    resource_version=resource.resource_version,
                    timeout_seconds=min(_WATCH_TIMEOUT_SECONDS, max(1, math.ceil(timeout)))
                    if timeout is not None
                    else _WATCH_TIMEOUT_SECONDS,
                    _stop_event=stop_event,
                ):
                    if stop_event.is_set():
                        return

                    if event_type == "BOOKMARK":
                        continue
                    assert isinstance(obj, self.kind), f"Expected {self.kind}, got {type(obj)}"
                    for name, predicate in predicates.items():
                        if predicate(event_type, obj):
                            result.predicate_name = name
                            return

            except Exception as e:
                logger.error(
                    "Error in wait_for watch: kind=%s namespace=%s name=%s error=%s",
                    resource.gvk().kind,
                    resource.namespace,
                    resource.name,
                    str(e),
                )
                result.error = e

        # Start the watch in a separate thread
        watch_thread = threading.Thread(
            target=copy_context().run, args=(watch_and_evaluate,), daemon=True
        )
        watch_thread.start()

        try:
            # Wait for the thread to complete or timeout
            watch_thread.join(timeout=timeout)

            # Handle timeout case
            if watch_thread.is_alive():
                raise WaitTimeout("Timeout waiting for condition")

            # Handle error case
            if result.error is not None:
                raise result.error

            # Handle unexpected termination
            if result.predicate_name is None:
                logger.error(
                    "Watch ended unexpectedly: kind=%s namespace=%s name=%s",
                    resource.gvk().kind,
                    resource.namespace,
                    resource.name,
                )
                raise RuntimeError("Watch ended unexpectedly")

            return result.predicate_name
        finally:
            # Ensure stop_event is set in case we hit an error before setting it
            stop_event.set()

    def scale(self, body: T, replicas: int) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata must be set for {body=}")
        if "scale" not in self.subresources:
            raise ValueError(f"Resource kind='{self.kind.gvk().kind}' does not support scale")
        namespace = body.namespace or self.default_namespace if self.namespaced else None
        name = body.name
        if not name:
            raise ValueError("name must be set for scale operation")
        url = f"{self._build_url(namespace=namespace, name=name)}/scale"
        # Match kubectl's simple patch format
        scale_body = {"spec": {"replicas": replicas}}
        response = self._client.patch(
            url, json=scale_body, headers={"Content-Type": "application/merge-patch+json"}
        )
        self._handle_scale_response(response, namespace or "", name)
        return self.get(name, namespace)

patch(body, operations, *, subresource=None, dry_run=False)

Apply an RFC 6902 JSON Patch; include tests for optimistic concurrency.

Source code in cloudcoil/client/_api_client.py
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
def patch(
    self,
    body: T,
    operations: list[dict[str, Any]],
    *,
    subresource: Literal["status"] | None = None,
    dry_run: bool = False,
) -> T:
    """Apply an RFC 6902 JSON Patch; include tests for optimistic concurrency."""
    url = self._patch_url(body, subresource)
    response = self._client.patch(
        url,
        json=operations,
        headers={"Content-Type": "application/json-patch+json"},
        params={"dryRun": "All"} if dry_run else {},
    )
    return self._handle_create_response(response)

AsyncAPIClient

Bases: _BaseAPIClient[T]

Source code in cloudcoil/client/_api_client.py
 606
 607
 608
 609
 610
 611
 612
 613
 614
 615
 616
 617
 618
 619
 620
 621
 622
 623
 624
 625
 626
 627
 628
 629
 630
 631
 632
 633
 634
 635
 636
 637
 638
 639
 640
 641
 642
 643
 644
 645
 646
 647
 648
 649
 650
 651
 652
 653
 654
 655
 656
 657
 658
 659
 660
 661
 662
 663
 664
 665
 666
 667
 668
 669
 670
 671
 672
 673
 674
 675
 676
 677
 678
 679
 680
 681
 682
 683
 684
 685
 686
 687
 688
 689
 690
 691
 692
 693
 694
 695
 696
 697
 698
 699
 700
 701
 702
 703
 704
 705
 706
 707
 708
 709
 710
 711
 712
 713
 714
 715
 716
 717
 718
 719
 720
 721
 722
 723
 724
 725
 726
 727
 728
 729
 730
 731
 732
 733
 734
 735
 736
 737
 738
 739
 740
 741
 742
 743
 744
 745
 746
 747
 748
 749
 750
 751
 752
 753
 754
 755
 756
 757
 758
 759
 760
 761
 762
 763
 764
 765
 766
 767
 768
 769
 770
 771
 772
 773
 774
 775
 776
 777
 778
 779
 780
 781
 782
 783
 784
 785
 786
 787
 788
 789
 790
 791
 792
 793
 794
 795
 796
 797
 798
 799
 800
 801
 802
 803
 804
 805
 806
 807
 808
 809
 810
 811
 812
 813
 814
 815
 816
 817
 818
 819
 820
 821
 822
 823
 824
 825
 826
 827
 828
 829
 830
 831
 832
 833
 834
 835
 836
 837
 838
 839
 840
 841
 842
 843
 844
 845
 846
 847
 848
 849
 850
 851
 852
 853
 854
 855
 856
 857
 858
 859
 860
 861
 862
 863
 864
 865
 866
 867
 868
 869
 870
 871
 872
 873
 874
 875
 876
 877
 878
 879
 880
 881
 882
 883
 884
 885
 886
 887
 888
 889
 890
 891
 892
 893
 894
 895
 896
 897
 898
 899
 900
 901
 902
 903
 904
 905
 906
 907
 908
 909
 910
 911
 912
 913
 914
 915
 916
 917
 918
 919
 920
 921
 922
 923
 924
 925
 926
 927
 928
 929
 930
 931
 932
 933
 934
 935
 936
 937
 938
 939
 940
 941
 942
 943
 944
 945
 946
 947
 948
 949
 950
 951
 952
 953
 954
 955
 956
 957
 958
 959
 960
 961
 962
 963
 964
 965
 966
 967
 968
 969
 970
 971
 972
 973
 974
 975
 976
 977
 978
 979
 980
 981
 982
 983
 984
 985
 986
 987
 988
 989
 990
 991
 992
 993
 994
 995
 996
 997
 998
 999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
class AsyncAPIClient(_BaseAPIClient[T]):
    def __init__(
        self,
        api_version: str,
        kind: Type[T],
        resource: str,
        subresources: list[str],
        default_namespace: str,
        namespaced: bool,
        client: httpx.AsyncClient,
    ) -> None:
        super().__init__(api_version, kind, resource, subresources, default_namespace, namespaced)
        self._client = client

    async def get(self, name: str, namespace: str | None = None) -> T:
        namespace = namespace or self.default_namespace
        url = self._build_url(name=name, namespace=namespace)
        response = await self._client.get(url)
        return self._handle_get_response(response, namespace, name)

    async def create(self, body: T, dry_run: bool = False) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata.name must be set for {body=}")
        namespace = body.namespace or self.default_namespace
        url = self._build_url(namespace=namespace)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        response = await self._client.post(
            url, json=body.model_dump(mode="json", by_alias=True, exclude_none=True), params=params
        )
        return self._handle_create_response(response)

    async def update(self, body: T, dry_run: bool = False) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata must be set for {body=}")
        namespace = body.namespace or self.default_namespace
        name = body.name
        if not name:
            raise ValueError("metadata.name must be set for update operations")
        url = self._build_url(namespace=namespace, name=name)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        response = await self._client.put(
            url, json=body.model_dump(mode="json", by_alias=True, exclude_none=True), params=params
        )
        return self._handle_create_response(response)

    async def patch(
        self,
        body: T,
        operations: list[dict[str, Any]],
        *,
        subresource: Literal["status"] | None = None,
        dry_run: bool = False,
    ) -> T:
        """Apply an RFC 6902 JSON Patch without blocking the event loop."""
        url = self._patch_url(body, subresource)
        response = await self._client.patch(
            url,
            json=operations,
            headers={"Content-Type": "application/json-patch+json"},
            params={"dryRun": "All"} if dry_run else {},
        )
        return self._handle_create_response(response)

    async def update_status(self, body: T, dry_run: bool = False) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata must be set for {body=}")
        if "status" not in self.subresources:
            raise ValueError(f"Resource {body.gvk().kind} does not support status updates")
        namespace = body.namespace or self.default_namespace
        name = body.name
        if not name:
            raise ValueError("metadata.name must be set for update operations")
        url = f"{self._build_url(namespace=namespace, name=name)}/status"
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        response = await self._client.put(
            url, json=body.model_dump(mode="json", by_alias=True, exclude_none=True), params=params
        )
        return self._handle_create_response(response)

    async def delete(
        self,
        name: str,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
        *,
        uid: str | None = None,
        resource_version: str | None = None,
    ) -> T | Status:
        namespace = namespace or self.default_namespace
        url = self._build_url(name=name, namespace=namespace)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        if propagation_policy:
            params["propagationPolicy"] = propagation_policy.capitalize()
        if grace_period_seconds is not None:
            params["gracePeriodSeconds"] = grace_period_seconds
        preconditions = {
            key: value
            for key, value in {
                "uid": uid,
                "resourceVersion": resource_version,
            }.items()
            if value is not None
        }
        if preconditions:
            response = await self._client.request(
                "DELETE",
                url,
                params=params,
                json={"apiVersion": "v1", "kind": "DeleteOptions", "preconditions": preconditions},
            )
        else:
            response = await self._client.delete(url, params=params)
        return self._handle_delete_response(response, namespace, name)

    async def remove(
        self,
        body: T,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
    ) -> T | Status:
        if not (body.metadata and body.metadata.name):
            raise ValueError(f"metadata.name must be set for {body=}")
        namespace = body.metadata.namespace or self.default_namespace
        name = body.metadata.name
        return await self.delete(
            name,
            namespace,
            dry_run=dry_run,
            propagation_policy=propagation_policy,
            grace_period_seconds=grace_period_seconds,
        )

    async def list(
        self,
        namespace: str | None = None,
        all_namespaces: bool = False,
        continue_: None | str = None,
        field_selector: str | None = None,
        label_selector: str | None = None,
        limit: int = DEFAULT_PAGE_LIMIT,
    ) -> ResourceList[T]:
        namespace = namespace or self.default_namespace
        if all_namespaces:
            namespace = None

        url = self._build_url(namespace=namespace)
        params: dict[str, str | int] = {}
        if continue_:
            params["continue"] = continue_
        if field_selector:
            params["fieldSelector"] = field_selector
        if label_selector:
            params["labelSelector"] = label_selector
        if limit:
            params["limit"] = limit
        response = await self._client.get(url, params=params)
        self._raise_for_status(response)
        output = ResourceList[self.kind].model_validate_json(response.content)  # type: ignore
        assert output.metadata
        output._page_client = self
        output._next_page_params = {
            "namespace": namespace,
            "all_namespaces": all_namespaces,
            "continue_": output.metadata.continue_,
            "field_selector": field_selector,
            "label_selector": label_selector,
            "limit": limit,
        }
        return output

    async def delete_all(
        self,
        namespace: str | None = None,
        dry_run: bool = False,
        propagation_policy: Literal["orphan", "background", "foreground"] | None = None,
        grace_period_seconds: int | None = None,
        label_selector: str | None = None,
        field_selector: str | None = None,
    ) -> ResourceList[T]:
        namespace = namespace or self.default_namespace
        url = self._build_url(namespace=namespace)
        params: dict[str, Any] = {}
        if dry_run:
            params["dryRun"] = "All"
        if propagation_policy:
            params["propagationPolicy"] = propagation_policy.capitalize()
        if grace_period_seconds is not None:
            params["gracePeriodSeconds"] = grace_period_seconds
        if label_selector:
            params["labelSelector"] = label_selector
        if field_selector:
            params["fieldSelector"] = field_selector
        logger.warning(
            "Deleting all resources: kind=%s namespace=%s policy=%s grace=%s label_selector=%s field_selector=%s",
            self.kind.gvk().kind,
            namespace,
            propagation_policy,
            grace_period_seconds,
            label_selector,
            field_selector,
        )
        response = await self._client.delete(url, params=params)
        self._raise_for_status(response)
        return ResourceList[self.kind].model_validate_json(response.content)  # type: ignore

    async def watch(
        self,
        namespace: str | None = None,
        all_namespaces: bool = False,
        field_selector: str | None = None,
        label_selector: str | None = None,
        resource_version: str | None = None,
        timeout_seconds: int | None = None,
        _raise_on_expired: bool = False,
    ) -> AsyncGenerator[tuple[WatchEvent, T] | tuple[BookmarkEvent, Unstructured], None]:
        retry_count = 0
        curr_resource_version = resource_version
        kind_name = self.kind.gvk().kind

        while True:
            try:
                url, params = self._build_watch_params(
                    namespace=namespace,
                    all_namespaces=all_namespaces,
                    field_selector=field_selector,
                    label_selector=label_selector,
                    resource_version=curr_resource_version,
                    timeout=timeout_seconds or _WATCH_TIMEOUT_SECONDS,
                )

                async with self._client.stream(
                    "GET",
                    url,
                    params=params,
                    timeout=(timeout_seconds or _WATCH_TIMEOUT_SECONDS) + 5,
                ) as response:
                    if response.status_code == 410:  # Gone
                        if _raise_on_expired:
                            raise WatchExpired(
                                "Watch history expired; relist required", status_code=410
                            )
                        logger.debug(
                            "Watch resource version expired, restarting: kind=%s namespace=%s",
                            kind_name,
                            namespace,
                        )
                        curr_resource_version = None
                        continue

                    response.raise_for_status()
                    retry_count = 0  # Reset retry counter on successful connection

                    async for line in response.aiter_lines():
                        if not line:
                            continue
                        event = json.loads(line)
                        type_ = event["type"]

                        if type_ == "ERROR":
                            if _raise_on_expired and (
                                event["object"].get("code") == 410
                                or event["object"].get("reason") in {"Expired", "Gone"}
                            ):
                                raise WatchExpired(
                                    "Watch history expired; relist required", status_code=410
                                )
                            code = event["object"].get("code")
                            if (
                                isinstance(code, int)
                                and 400 <= code < 500
                                and code not in {408, 410, 429}
                            ):
                                raise APIError(event["object"], status_code=code)
                            if "status" in event["object"]:
                                status = event["object"]["status"]
                                if status == "Failure":
                                    reason = event["object"].get("reason", "")
                                    if reason == "Expired":
                                        logger.debug(
                                            "Watch resource version expired, restarting: kind=%s namespace=%s",
                                            kind_name,
                                            namespace,
                                        )
                                        curr_resource_version = None
                                        break
                            logger.error("Watch error event received: %s", event)
                            raise WatchError(f"Watch error: {event}")

                        # Handle bookmark events specially - they only contain minimal data
                        if type_ == "BOOKMARK":
                            # For bookmark events, create an Unstructured object to avoid validation errors
                            bookmark_obj: Unstructured = Unstructured.model_validate(
                                event["object"]
                            )
                            if bookmark_obj.metadata and bookmark_obj.metadata.resource_version:
                                curr_resource_version = bookmark_obj.metadata.resource_version
                            yield type_, bookmark_obj
                        else:
                            obj = self.kind.model_validate(event["object"])
                            if obj.metadata and obj.metadata.resource_version:
                                curr_resource_version = obj.metadata.resource_version
                            yield type_, obj

            except WatchExpired:
                raise
            except (httpx.RequestError, httpx.HTTPStatusError, WatchError) as e:
                if isinstance(e, httpx.HTTPStatusError):
                    if e.response.status_code == 410:  # Gone
                        logger.debug(
                            "Watch resource version expired, restarting: kind=%s namespace=%s",
                            self.kind.gvk().kind,
                            namespace,
                        )
                        curr_resource_version = None
                        continue
                    if e.response.status_code == 404:  # Not Found
                        logger.error(
                            "Watch endpoint not found: kind=%s namespace=%s",
                            self.kind.gvk().kind,
                            namespace,
                        )
                        raise ResourceNotFound("Watch endpoint not found", status_code=404) from e
                    if 400 <= e.response.status_code < 500 and e.response.status_code not in {
                        408,
                        429,
                    }:
                        raise APIError(str(e), status_code=e.response.status_code) from e

                retry_count += 1
                backoff = self._get_backoff_time(retry_count)
                logger.warning(
                    "Watch connection failed, retrying in %0.1fs: %s",
                    backoff,
                    str(e),
                )
                await asyncio.sleep(backoff)

    async def wait_for(
        self,
        resource: T,
        predicates: dict[str, Callable[[WatchEvent, T], bool | None]],
        timeout: float | None = None,
    ) -> str:
        """Async version of wait_for that uses asyncio for timing out the watch."""

        if not resource.name:
            raise ValueError("metadata.name must be set to wait for a resource")
        if not predicates:
            raise ValueError("At least one wait predicate is required")
        if timeout is not None and timeout <= 0:
            raise WaitTimeout("Timeout waiting for condition")

        async def watch_and_evaluate() -> str:
            async with aclosing(
                self.watch(
                    namespace=resource.namespace,
                    field_selector=f"metadata.name={resource.name}",
                    resource_version=resource.resource_version,
                )
            ) as stream:
                async for event_type, obj in stream:
                    if event_type == "BOOKMARK":
                        continue

                    assert isinstance(obj, self.kind), f"Expected {self.kind}, got {type(obj)}"
                    for name, predicate in predicates.items():
                        result = predicate(event_type, obj)
                        if result:
                            return name
            raise RuntimeError("Watch ended unexpectedly")

        try:
            return await asyncio.wait_for(watch_and_evaluate(), timeout=timeout)
        except asyncio.TimeoutError:
            raise WaitTimeout("Timeout waiting for condition")

    async def scale(self, body: T, replicas: int) -> T:
        if not (body.metadata):
            raise ValueError(f"metadata must be set for {body=}")
        if "scale" not in self.subresources:
            raise ValueError(f"Resource kind='{self.kind.gvk().kind}' does not support scale")
        namespace = body.namespace or self.default_namespace if self.namespaced else None
        name = body.name
        if not name:
            raise ValueError("name must be set for scale operation")
        url = f"{self._build_url(namespace=namespace, name=name)}/scale"
        # Match kubectl's simple patch format
        scale_body = {"spec": {"replicas": replicas}}
        response = await self._client.patch(
            url, json=scale_body, headers={"Content-Type": "application/merge-patch+json"}
        )
        self._handle_scale_response(response, namespace or "", name)
        return await self.get(name, namespace)

patch(body, operations, *, subresource=None, dry_run=False) async

Apply an RFC 6902 JSON Patch without blocking the event loop.

Source code in cloudcoil/client/_api_client.py
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
async def patch(
    self,
    body: T,
    operations: list[dict[str, Any]],
    *,
    subresource: Literal["status"] | None = None,
    dry_run: bool = False,
) -> T:
    """Apply an RFC 6902 JSON Patch without blocking the event loop."""
    url = self._patch_url(body, subresource)
    response = await self._client.patch(
        url,
        json=operations,
        headers={"Content-Type": "application/json-patch+json"},
        params={"dryRun": "All"} if dry_run else {},
    )
    return self._handle_create_response(response)

wait_for(resource, predicates, timeout=None) async

Async version of wait_for that uses asyncio for timing out the watch.

Source code in cloudcoil/client/_api_client.py
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
async def wait_for(
    self,
    resource: T,
    predicates: dict[str, Callable[[WatchEvent, T], bool | None]],
    timeout: float | None = None,
) -> str:
    """Async version of wait_for that uses asyncio for timing out the watch."""

    if not resource.name:
        raise ValueError("metadata.name must be set to wait for a resource")
    if not predicates:
        raise ValueError("At least one wait predicate is required")
    if timeout is not None and timeout <= 0:
        raise WaitTimeout("Timeout waiting for condition")

    async def watch_and_evaluate() -> str:
        async with aclosing(
            self.watch(
                namespace=resource.namespace,
                field_selector=f"metadata.name={resource.name}",
                resource_version=resource.resource_version,
            )
        ) as stream:
            async for event_type, obj in stream:
                if event_type == "BOOKMARK":
                    continue

                assert isinstance(obj, self.kind), f"Expected {self.kind}, got {type(obj)}"
                for name, predicate in predicates.items():
                    result = predicate(event_type, obj)
                    if result:
                        return name
        raise RuntimeError("Watch ended unexpectedly")

    try:
        return await asyncio.wait_for(watch_and_evaluate(), timeout=timeout)
    except asyncio.TimeoutError:
        raise WaitTimeout("Timeout waiting for condition")

Controllers

Typed asynchronous Kubernetes reconciliation and controller lifecycle.

Controller

Reconcile one primary kind, with optional child and dependency watches.

Configure watches before run(). Each instance runs once. The runtime owns its informers and worker tasks; Config ownership remains with the caller. Workers run only after every informer has synced. Failures retry with capped backoff; cancellation and fatal watch errors stop the controller and its sibling tasks.

Source code in cloudcoil/controller/_controller.py
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
class Controller[T: Resource]:
    """Reconcile one primary kind, with optional child and dependency watches.

    Configure watches before run(). Each instance runs once. The runtime owns its
    informers and worker tasks; Config ownership remains with the caller. Workers
    run only after every informer has synced. Failures retry with capped backoff;
    cancellation and fatal watch errors stop the controller and its sibling tasks.
    """

    def __init__(
        self,
        resource: type[T],
        reconcile: Reconciler[T] | None = None,
        *,
        name: str | None = None,
        owns: tuple[type[Resource], ...] = (),
        report_status: bool | None = None,
        config: Config | None = None,
        namespace: str | None = None,
        all_namespaces: bool = False,
        label_selector: str | None = None,
        workers: int = 1,
        resync_period: float = 300,
        sync_timeout: float = 30,
        shutdown_timeout: float = 10,
        reconcile_timeout: float | None = None,
        events: bool | EventRecorder = True,
        status_updates: bool | None = None,
        event_flush_timeout: float = 2,
    ) -> None:
        if isinstance(workers, bool) or not isinstance(workers, int) or workers < 1:
            raise ValueError("workers must be a positive integer")
        for setting, value in (
            ("sync_timeout", sync_timeout),
            ("shutdown_timeout", shutdown_timeout),
            ("reconcile_timeout", reconcile_timeout),
            ("event_flush_timeout", event_flush_timeout),
        ):
            if value is not None and (not math.isfinite(value) or value <= 0):
                raise ValueError(f"{setting} must be finite and positive")
        if name is not None and not name.strip():
            raise ValueError("Controller name must not be empty")
        if name is not None and len(name) > 118:
            raise ValueError("Controller name must have at most 118 characters")
        self.name = name
        if not isinstance(events, (bool, EventRecorder)):
            raise TypeError("events must be a bool or EventRecorder")
        if status_updates is not None and not isinstance(status_updates, bool):
            raise TypeError("status_updates must be a bool")
        self._registry = Registry(resource, report_status)
        self._admission_registry = AdmissionRegistry(AdmissionWebhook(), self._registry.check)
        self._status_updates = (
            status_updates
            if status_updates is not None
            else not (
                isinstance(reconcile, (Stages, Cases))
                or (reconcile is None and self._registry.report_status)
            )
        )
        self._events = (
            events
            if isinstance(events, EventRecorder)
            else EventRecorder(f"cloudcoil/{name or resource.gvk().kind.lower()}")
            if events
            else None
        )
        self._metrics = _ReconcileMetrics()
        self.resource = resource
        self._reconcile = reconcile
        self.config = config
        self._options = InformerOptions(
            namespace=namespace,
            all_namespaces=all_namespaces,
            label_selector=label_selector,
            resync_period=resync_period,
            max_items=0,
        )
        self._workers = workers
        self._sync_timeout = sync_timeout
        self._shutdown_timeout = shutdown_timeout
        self._reconcile_timeout = reconcile_timeout
        self._event_flush_timeout = event_flush_timeout
        self._queue = WorkQueue[ResourceKey]()
        self._watches: list[_Watch] = []
        self._informers: list[AsyncInformer[Any]] = []
        self._readers: dict[type[Resource], AsyncInformer[Any]] = {}
        self._primary: AsyncInformer[T] | None = None
        self._primary_namespaced = True
        self._pool: _InformerPool | None = None
        self._prepared_config: Config | None = None
        self._used = False
        self._ready = asyncio.Event()
        self._finished = asyncio.Event()
        self._failure: BaseException | None = None
        self.owns(*owns)

    @overload
    def reconcile(self, request: Request[T]) -> Awaitable[T | Result | Wait | None]: ...

    @overload
    def reconcile[F: Callable[..., Any]](
        self, *, every: float | None = None
    ) -> Callable[[F], F]: ...

    def reconcile(self, request: Request[T] | None = None, *, every: float | None = None) -> Any:
        """Register an async (resource, optional ctx) handler.

        Passing Request explicitly executes a reconciliation for low-level embedding.
        """
        if request is not None:
            return self._invoke(request)
        if self._reconcile is not None:
            raise ValueError("A constructor reconciler is already registered")
        return self._registry.reconcile(every=every)

    def stage(
        self,
        name: str | None = None,
        *,
        condition: str,
        depends: object | tuple[object, ...] | None = None,
    ) -> StageScope[T]:
        return self._registry.stage(name, condition=condition, depends=depends)

    def case[F: Callable[..., Any]](
        self,
        *,
        when: Callable[..., bool],
        after: object | tuple[object, ...] | None = None,
    ) -> Callable[[F], F]:
        return self._registry.branches.case(when=when, after=after)

    def otherwise[F: Callable[..., Any]](self) -> Callable[[F], F]:
        return self._registry.branches.otherwise()

    def finalize[F: Callable[..., Any]](self, key: str) -> Callable[[F], F]:
        return self._registry.finalize(key)

    def validate(self, **options: Any) -> Callable[[Validator[T]], Validator[T]]:
        """Register admission validation for this controller's resource."""
        return self._admission_registry.validate(self.resource, **options)

    def mutate(self, **options: Any) -> Callable[[Mutator[T]], Mutator[T]]:
        """Register admission mutation returning the edited resource."""
        return self._admission_registry.mutate(self.resource, **options)

    def _validate(self, *, freeze: bool = False) -> None:
        if self._reconcile is None:
            self._registry.validate(freeze=freeze)
        elif (
            self._registry.handler
            or self._registry.stages
            or self._registry.branches.populated
            or self._registry.finalizer
        ):
            raise ValueError(
                "Constructor reconciliation cannot be combined with decorated handlers"
            )
        elif isinstance(self._reconcile, (Stages, Cases)):
            self._reconcile._freeze(self.resource)
        if freeze:
            self._registry.frozen = True

    async def _invoke(self, request: Request[T]) -> T | Result | Wait | None:
        if self._reconcile is not None:
            return await self._reconcile(request)
        return await self._registry(request)

    def owns(self, *resources: type[Resource]) -> Self:
        """Enqueue primary owners when a child changes (direct controller references).

        Owner matching uses group/kind and UID, including across served versions.
        For indirect or non-owning relationships use watch(..., mapper=...).
        Application manifests grant get/list/watch/create/patch for owned children.
        Deletion is left to Kubernetes garbage collection or explicit RBAC.
        """
        for resource in resources:
            self._add_watch(_Watch(resource))
        return self

    @overload
    def watch[U: Resource](self, resource: type[U], *, mapper: Mapper[U]) -> Self: ...

    @overload
    def watch[U: Resource](self, resource: type[U]) -> Callable[[Mapper[U]], Mapper[U]]: ...

    def watch[U: Resource](
        self,
        resource: type[U],
        *,
        mapper: Mapper[U] | None = None,
    ) -> Self | Callable[[Mapper[U]], Mapper[U]]:
        """Register a synchronous dependency mapper; updates map old and new state."""
        if mapper is not None:
            return self._add_watch(_Watch(resource, mapper))

        def register(handler: Mapper[U]) -> Mapper[U]:
            import inspect

            if inspect.iscoroutinefunction(handler):
                raise TypeError("Watch mappers must be synchronous")
            self._add_watch(_Watch(resource, handler))
            return handler

        return register

    def _add_watch(self, watch: _Watch) -> Self:
        if self._used or self._registry.frozen:
            raise RuntimeError("Configure watches before running the controller")
        self._watches.append(watch)
        return self

    def enqueue(self, key: ResourceKey) -> None:
        """Request a primary key explicitly, including from an external event source."""
        if not isinstance(key, ResourceKey):
            raise TypeError("enqueue expects a ResourceKey")
        self._queue.add(key)

    def cached[U: Resource](self, resource: type[U]) -> CachedResources[U]:
        """Read a declared informer, including from a secondary-event mapper.

        The primary informer syncs before secondary mappers run. Use explicit
        namespaces when mapping all-namespace dependencies. No new watch starts.
        """
        if resource not in self._readers:
            raise ValueError(f"{resource.__name__} is not initialized; declare a watch first")
        namespace = (
            self._options.namespace
            or (self._prepared_config or self.config or context.active_config).namespace
        )
        return CachedResources(cast(AsyncInformer[U], self._readers[resource]), namespace)

    @property
    def ready(self) -> bool:
        """Whether all watches have synced and reconcile workers are running."""
        return self._ready.is_set()

    @property
    def status(self) -> ControllerStatus:
        """Return local queue and reconcile statistics without inspecting cached objects."""
        counts = self._metrics.outcomes
        return ControllerStatus(
            ready=self.ready,
            queued=self._queue.depth,
            processing=self._queue.processing,
            delayed=self._queue.delayed,
            successes=counts["success"],
            errors=counts["error"],
            terminal_errors=counts["terminal"],
            cancellations=counts["cancelled"],
            duration_seconds=self._metrics.duration,
        )

    async def wait_ready(self, timeout: float = 30) -> None:
        """Wait for startup, propagating startup failure instead of hanging."""
        ready = asyncio.create_task(self._ready.wait())
        finished = asyncio.create_task(self._finished.wait())
        try:
            async with asyncio.timeout(timeout):
                await asyncio.wait((ready, finished), return_when=asyncio.FIRST_COMPLETED)
            if self._failure is not None:
                raise self._failure
            if not self.ready:
                raise RuntimeError("Controller stopped before becoming ready")
        finally:
            for task in (ready, finished):
                task.cancel()
            await asyncio.gather(ready, finished, return_exceptions=True)

    async def _enqueue_primary(self, obj: T) -> None:
        self.enqueue(ResourceKey.from_resource(obj))

    async def _update_primary(self, old: T | None, new: T) -> None:
        if (
            not self._status_updates
            and old is not None
            and old.resource_version != new.resource_version
        ):
            before = old.model_dump(mode="json", by_alias=True, exclude_none=True)
            after = new.model_dump(mode="json", by_alias=True, exclude_none=True)
            for value in (before, after):
                value.pop("status", None)
                metadata = value.get("metadata", {})
                metadata.pop("resourceVersion", None)
                metadata.pop("managedFields", None)
            if before == after:
                return
        # Same-version resyncs still run; metadata/deletion/spec and child events
        # remain inputs even when the controller ignores primary status changes.
        await self._enqueue_primary(new)

    def _owner_keys(self, obj: Resource) -> Iterable[ResourceKey]:
        if not obj.metadata or self._primary is None:
            return
        primary_gvk = self.resource.gvk()
        namespace = obj.namespace if self._primary_namespaced else None
        for ref in obj.metadata.owner_references or []:
            if not ref.controller or ref.kind != primary_gvk.kind:
                continue
            if ref.api_version.rpartition("/")[0] != primary_gvk.group:
                continue
            owner = self._primary.get(ref.name, namespace)
            if owner is not None and owner.metadata and owner.metadata.uid == ref.uid:
                yield ResourceKey(ref.name, namespace)

    async def _install(self, config: Config) -> None:
        self._validate(freeze=True)
        primary_client = await config.async_client_for(self.resource, cached=False)
        self._primary_namespaced = primary_client.namespaced
        self._primary = (
            self._pool.get(config, primary_client, self._options)
            if self._pool is not None
            else AsyncInformer(primary_client, self._options)
        )
        self._primary.on_add(self._enqueue_primary)
        self._primary.on_update(self._update_primary)
        self._primary.on_delete(self._enqueue_primary)
        self._informers.append(self._primary)
        self._readers[self.resource] = self._primary
        for watch in self._watches:
            client = await config.async_client_for(watch.resource, cached=False)
            # A primary selector is not generally a selector for its dependencies.
            options = self._options.model_copy(update={"label_selector": None})
            # Cluster-scoped owners may have children in any namespace.
            if not self._primary_namespaced and client.namespaced:
                options = options.model_copy(update={"namespace": None, "all_namespaces": True})
            informer = (
                self._pool.get(config, client, options)
                if self._pool is not None
                else AsyncInformer(client, options)
            )
            mapper = watch.mapper or self._owner_keys

            async def changed(obj: Resource, mapper: Mapper[Any] = mapper) -> None:
                try:
                    # Validate the entire mapping before enqueueing a partial result.
                    keys = list(mapper(obj.model_copy(deep=True)))
                    if not all(isinstance(key, ResourceKey) for key in keys):
                        raise TypeError("Watch mappers must return ResourceKey values")
                    for key in keys:
                        self.enqueue(key)
                except Exception as exc:
                    self._failure = exc
                    self._mapping_failed.set()

            async def updated(old: Resource | None, new: Resource, changed: Any = changed) -> None:
                if old is not None:
                    await changed(old)
                await changed(new)

            informer.on_add(changed)
            informer.on_delete(changed)
            informer.on_update(updated)
            self._informers.append(informer)
            # Primary selectors may differ from secondary watches of the same kind.
            # cached(Primary) consistently refers to the primary snapshot.
            self._readers.setdefault(watch.resource, informer)

    async def _sync(self) -> None:
        async with asyncio.timeout(self._sync_timeout):
            # Primary sync first makes owner UID checks meaningful for secondary lists.
            for informer in self._informers:
                await informer._start()
                assert informer._watch._task is not None
                waiter = asyncio.create_task(informer._sync_event.wait())
                try:
                    watches = [
                        item._watch._task
                        for item in self._informers
                        if item._watch._task is not None
                    ]
                    done, _ = await asyncio.wait(
                        (waiter, *watches), return_when=asyncio.FIRST_COMPLETED
                    )
                    for item in self._informers:
                        if item._watch._task in done:
                            raise item._watch._error or RuntimeError("Informer stopped before sync")
                finally:
                    waiter.cancel()
                    await asyncio.gather(waiter, return_exceptions=True)

    async def _flush_events(self, request: Request[T]) -> None:
        if self._events is None or request.resource is None:
            return
        try:
            # Event I/O never consumes the reconcile deadline or changes its outcome.
            async with asyncio.timeout(self._event_flush_timeout):
                for reason, message, event_type, action in request._report.events:
                    await self._events.emit(
                        request.object,
                        reason,
                        message,
                        type=event_type,
                        action=action,
                        config=request.config,
                    )
        except Exception:
            logger.warning("Could not flush reconciliation Events", exc_info=True)

    async def _report_failure(
        self, original: T | None, request: Request[T], error: Exception
    ) -> bool:
        try:
            async with asyncio.timeout(5):
                if request._report.baseline is not None:
                    original = cast(T, request._report.baseline)
                if request._report.managed and request._report.dirty:
                    request.object.status = request._report.status.model_copy(deep=True)  # type: ignore[attr-defined]
                request._failed(error)
                if original is not None and request._report.dirty:
                    desired = original.model_copy(deep=True)
                    desired.status = request._report.status  # type: ignore[attr-defined]
                    await _persist(original, desired)
            await self._flush_events(request)
            return True
        except Exception:
            # Report conflicts/errors must retry even if the business error was terminal.
            logger.exception("Could not persist failure status for %s", request.key)
            return False

    async def _worker(self) -> None:
        assert self._primary is not None
        while True:
            try:
                key = await self._queue.get()
            except QueueClosed:
                return
            started = time.monotonic()
            outcome = "success"
            request: Request[T] | None = None
            original: T | None = None
            callback_complete = False
            try:
                resource = self._primary.get(key.name, key.namespace)
                # Keep an independent dispatch baseline even if the informer changes
                # while reconcile awaits, or the caller edits request.resource in place.
                original = resource.model_copy(deep=True) if resource is not None else None
                request = Request(
                    key,
                    original.model_copy(deep=True) if original is not None else None,
                    config=context.active_config,
                    _informers=self._readers,
                    _events=self._events,
                )
                async with asyncio.timeout(self._reconcile_timeout):
                    try:
                        returned = await self._invoke(request)
                    except Wait as wait:
                        returned = wait
                    if request._report.baseline is not None:
                        original = cast(T, request._report.baseline)
                    callback_complete = True
                    result = (
                        Result(resource=returned) if isinstance(returned, Resource) else returned
                    )
                    if isinstance(result, Wait):
                        result = Result(requeue_after=result.requeue_after)
                    if result is not None and not isinstance(result, Result):
                        raise TypeError("Reconcile must return its resource, Result, Wait, or None")
                    if request._report.dirty and (result is None or result.resource is None):
                        assert original is not None
                        desired = original.model_copy(deep=True)
                        desired.status = request._report.status  # type: ignore[attr-defined]
                        result = Result(
                            resource=desired, requeue_after=result.requeue_after if result else None
                        )
                    if result is not None and result.resource is not None:
                        if original is None:
                            raise ValueError(
                                "Cannot persist a returned resource for an absent snapshot"
                            )
                        if not isinstance(result.resource, self.resource):
                            raise TypeError("Reconcile must return its primary resource type")
                        await _persist(original, result.resource)
                self._queue.forget(key)
                if result is not None and result.requeue_after is not None:
                    self._queue.add_after(key, result.requeue_after)
                await self._flush_events(request)
            except asyncio.CancelledError:
                outcome = "cancelled"
                raise
            except TerminalError as error:
                outcome = "terminal"
                reported = (
                    await self._report_failure(original, request, error)
                    if request is not None and not callback_complete
                    else True
                )
                if reported:
                    self._queue.forget(key)
                else:
                    outcome = "error"
                    self._queue.retry(key)
                logger.exception(
                    "Terminal reconcile error for %s %s", self.resource.gvk().kind, key
                )
            except Exception as error:
                outcome = "error"
                if request is not None and not callback_complete:
                    await self._report_failure(original, request, error)
                delay = self._queue.retry(key)
                logger.exception(
                    "Reconcile failed for %s %s; retry in %.2fs",
                    self.resource.gvk().kind,
                    key,
                    delay,
                )
            finally:
                self._metrics.observe(outcome, time.monotonic() - started)
                self._queue.done(key)

    async def run(self, *, stop: asyncio.Event | None = None) -> None:
        """Run until stop is set or the caller cancels; drain on an explicit stop.

        An explicit stop drains accepted ready work up to shutdown_timeout, then
        cancels remaining workers. Cancellation/fatal errors cancel workers directly.
        Delayed retries are discarded; the next process recovers by listing state.
        """
        if self._used:
            raise RuntimeError("Controller instances can only run once")
        self._used = True
        stop = stop if stop is not None else asyncio.Event()
        self._mapping_failed = asyncio.Event()
        tasks: list[asyncio.Task[Any]] = []
        workers: list[asyncio.Task[None]] = []
        graceful = False
        try:
            config = self._prepared_config or self.config or context.active_config
            async with config:
                try:
                    if self._prepared_config is None:
                        await self._install(config)
                    sync = asyncio.create_task(self._sync())
                    stopped = asyncio.create_task(stop.wait())
                    mapping_failed = asyncio.create_task(self._mapping_failed.wait())
                    tasks.extend((sync, stopped, mapping_failed))
                    done, _ = await asyncio.wait(tasks, return_when=asyncio.FIRST_COMPLETED)
                    if stopped in done:
                        graceful = True
                        return
                    if mapping_failed in done:
                        raise self._failure or RuntimeError("Watch mapping failed")
                    await sync
                    workers = [asyncio.create_task(self._worker()) for _ in range(self._workers)]
                    tasks.extend(workers)
                    self._ready.set()
                    watches = [
                        i._watch._task for i in self._informers if i._watch._task is not None
                    ]
                    done, _ = await asyncio.wait(
                        [stopped, mapping_failed, *workers, *watches],
                        return_when=asyncio.FIRST_COMPLETED,
                    )
                    if mapping_failed in done:
                        raise self._failure or RuntimeError("Watch mapping failed")
                    if stopped in done:
                        graceful = True
                    else:
                        for informer in self._informers:
                            if informer._watch._error is not None:
                                raise informer._watch._error
                        for worker in workers:
                            if worker in done:
                                await worker
                        raise RuntimeError("Controller task stopped unexpectedly")
                finally:
                    self._ready.clear()
                    # Stop startup before stopping informers it might still create.
                    for task in tasks:
                        if task not in workers:
                            task.cancel()
                    await asyncio.gather(
                        *(task for task in tasks if task not in workers), return_exceptions=True
                    )
                    if self._pool is None:
                        for informer in reversed(self._informers):
                            await informer._stop()
                    self._queue.shutdown(immediate=not graceful)
                    if graceful and workers:
                        try:
                            async with asyncio.timeout(self._shutdown_timeout):
                                await asyncio.gather(*workers)
                        except TimeoutError:
                            pass
                    for worker in workers:
                        worker.cancel()
                    await asyncio.gather(*workers, return_exceptions=True)
        except BaseException as exc:
            self._failure = exc
            raise
        finally:
            self._queue.shutdown(immediate=True)
            self._finished.set()

ready property

Whether all watches have synced and reconcile workers are running.

status property

Return local queue and reconcile statistics without inspecting cached objects.

cached(resource)

Read a declared informer, including from a secondary-event mapper.

The primary informer syncs before secondary mappers run. Use explicit namespaces when mapping all-namespace dependencies. No new watch starts.

Source code in cloudcoil/controller/_controller.py
260
261
262
263
264
265
266
267
268
269
270
271
272
def cached[U: Resource](self, resource: type[U]) -> CachedResources[U]:
    """Read a declared informer, including from a secondary-event mapper.

    The primary informer syncs before secondary mappers run. Use explicit
    namespaces when mapping all-namespace dependencies. No new watch starts.
    """
    if resource not in self._readers:
        raise ValueError(f"{resource.__name__} is not initialized; declare a watch first")
    namespace = (
        self._options.namespace
        or (self._prepared_config or self.config or context.active_config).namespace
    )
    return CachedResources(cast(AsyncInformer[U], self._readers[resource]), namespace)

enqueue(key)

Request a primary key explicitly, including from an external event source.

Source code in cloudcoil/controller/_controller.py
254
255
256
257
258
def enqueue(self, key: ResourceKey) -> None:
    """Request a primary key explicitly, including from an external event source."""
    if not isinstance(key, ResourceKey):
        raise TypeError("enqueue expects a ResourceKey")
    self._queue.add(key)

mutate(**options)

Register admission mutation returning the edited resource.

Source code in cloudcoil/controller/_controller.py
184
185
186
def mutate(self, **options: Any) -> Callable[[Mutator[T]], Mutator[T]]:
    """Register admission mutation returning the edited resource."""
    return self._admission_registry.mutate(self.resource, **options)

owns(*resources)

Enqueue primary owners when a child changes (direct controller references).

Owner matching uses group/kind and UID, including across served versions. For indirect or non-owning relationships use watch(..., mapper=...). Application manifests grant get/list/watch/create/patch for owned children. Deletion is left to Kubernetes garbage collection or explicit RBAC.

Source code in cloudcoil/controller/_controller.py
210
211
212
213
214
215
216
217
218
219
220
def owns(self, *resources: type[Resource]) -> Self:
    """Enqueue primary owners when a child changes (direct controller references).

    Owner matching uses group/kind and UID, including across served versions.
    For indirect or non-owning relationships use watch(..., mapper=...).
    Application manifests grant get/list/watch/create/patch for owned children.
    Deletion is left to Kubernetes garbage collection or explicit RBAC.
    """
    for resource in resources:
        self._add_watch(_Watch(resource))
    return self

reconcile(request=None, *, every=None)

reconcile(
    request: Request[T],
) -> Awaitable[T | Result | Wait | None]
reconcile(
    *, every: float | None = None
) -> Callable[[F], F]

Register an async (resource, optional ctx) handler.

Passing Request explicitly executes a reconciliation for low-level embedding.

Source code in cloudcoil/controller/_controller.py
146
147
148
149
150
151
152
153
154
155
def reconcile(self, request: Request[T] | None = None, *, every: float | None = None) -> Any:
    """Register an async (resource, optional ctx) handler.

    Passing Request explicitly executes a reconciliation for low-level embedding.
    """
    if request is not None:
        return self._invoke(request)
    if self._reconcile is not None:
        raise ValueError("A constructor reconciler is already registered")
    return self._registry.reconcile(every=every)

run(*, stop=None) async

Run until stop is set or the caller cancels; drain on an explicit stop.

An explicit stop drains accepted ready work up to shutdown_timeout, then cancels remaining workers. Cancellation/fatal errors cancel workers directly. Delayed retries are discarded; the next process recovers by listing state.

Source code in cloudcoil/controller/_controller.py
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
async def run(self, *, stop: asyncio.Event | None = None) -> None:
    """Run until stop is set or the caller cancels; drain on an explicit stop.

    An explicit stop drains accepted ready work up to shutdown_timeout, then
    cancels remaining workers. Cancellation/fatal errors cancel workers directly.
    Delayed retries are discarded; the next process recovers by listing state.
    """
    if self._used:
        raise RuntimeError("Controller instances can only run once")
    self._used = True
    stop = stop if stop is not None else asyncio.Event()
    self._mapping_failed = asyncio.Event()
    tasks: list[asyncio.Task[Any]] = []
    workers: list[asyncio.Task[None]] = []
    graceful = False
    try:
        config = self._prepared_config or self.config or context.active_config
        async with config:
            try:
                if self._prepared_config is None:
                    await self._install(config)
                sync = asyncio.create_task(self._sync())
                stopped = asyncio.create_task(stop.wait())
                mapping_failed = asyncio.create_task(self._mapping_failed.wait())
                tasks.extend((sync, stopped, mapping_failed))
                done, _ = await asyncio.wait(tasks, return_when=asyncio.FIRST_COMPLETED)
                if stopped in done:
                    graceful = True
                    return
                if mapping_failed in done:
                    raise self._failure or RuntimeError("Watch mapping failed")
                await sync
                workers = [asyncio.create_task(self._worker()) for _ in range(self._workers)]
                tasks.extend(workers)
                self._ready.set()
                watches = [
                    i._watch._task for i in self._informers if i._watch._task is not None
                ]
                done, _ = await asyncio.wait(
                    [stopped, mapping_failed, *workers, *watches],
                    return_when=asyncio.FIRST_COMPLETED,
                )
                if mapping_failed in done:
                    raise self._failure or RuntimeError("Watch mapping failed")
                if stopped in done:
                    graceful = True
                else:
                    for informer in self._informers:
                        if informer._watch._error is not None:
                            raise informer._watch._error
                    for worker in workers:
                        if worker in done:
                            await worker
                    raise RuntimeError("Controller task stopped unexpectedly")
            finally:
                self._ready.clear()
                # Stop startup before stopping informers it might still create.
                for task in tasks:
                    if task not in workers:
                        task.cancel()
                await asyncio.gather(
                    *(task for task in tasks if task not in workers), return_exceptions=True
                )
                if self._pool is None:
                    for informer in reversed(self._informers):
                        await informer._stop()
                self._queue.shutdown(immediate=not graceful)
                if graceful and workers:
                    try:
                        async with asyncio.timeout(self._shutdown_timeout):
                            await asyncio.gather(*workers)
                    except TimeoutError:
                        pass
                for worker in workers:
                    worker.cancel()
                await asyncio.gather(*workers, return_exceptions=True)
    except BaseException as exc:
        self._failure = exc
        raise
    finally:
        self._queue.shutdown(immediate=True)
        self._finished.set()

validate(**options)

Register admission validation for this controller's resource.

Source code in cloudcoil/controller/_controller.py
180
181
182
def validate(self, **options: Any) -> Callable[[Validator[T]], Validator[T]]:
    """Register admission validation for this controller's resource."""
    return self._admission_registry.validate(self.resource, **options)

wait_ready(timeout=30) async

Wait for startup, propagating startup failure instead of hanging.

Source code in cloudcoil/controller/_controller.py
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
async def wait_ready(self, timeout: float = 30) -> None:
    """Wait for startup, propagating startup failure instead of hanging."""
    ready = asyncio.create_task(self._ready.wait())
    finished = asyncio.create_task(self._finished.wait())
    try:
        async with asyncio.timeout(timeout):
            await asyncio.wait((ready, finished), return_when=asyncio.FIRST_COMPLETED)
        if self._failure is not None:
            raise self._failure
        if not self.ready:
            raise RuntimeError("Controller stopped before becoming ready")
    finally:
        for task in (ready, finished):
            task.cancel()
        await asyncio.gather(ready, finished, return_exceptions=True)

watch(resource, *, mapper=None)

watch(resource: type[U], *, mapper: Mapper[U]) -> Self
watch(
    resource: type[U],
) -> Callable[[Mapper[U]], Mapper[U]]

Register a synchronous dependency mapper; updates map old and new state.

Source code in cloudcoil/controller/_controller.py
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
def watch[U: Resource](
    self,
    resource: type[U],
    *,
    mapper: Mapper[U] | None = None,
) -> Self | Callable[[Mapper[U]], Mapper[U]]:
    """Register a synchronous dependency mapper; updates map old and new state."""
    if mapper is not None:
        return self._add_watch(_Watch(resource, mapper))

    def register(handler: Mapper[U]) -> Mapper[U]:
        import inspect

        if inspect.iscoroutinefunction(handler):
            raise TypeError("Watch mappers must be synchronous")
        self._add_watch(_Watch(resource, handler))
        return handler

    return register

Context

Explicit clients and reporting for one primary resource.

Status and Events are queued; ensure and live clients perform I/O immediately. Context instances belong to one pass and must not be stored on the controller.

Source code in cloudcoil/controller/_context.py
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
class Context[T: Resource]:
    """Explicit clients and reporting for one primary resource.

    Status and Events are queued; ensure and live clients perform I/O immediately.
    Context instances belong to one pass and must not be stored on the controller.
    """

    def __init__(self, request: Request[T]) -> None:
        self._request = request

    @property
    def resource(self) -> T:
        return self._request.object

    @property
    def namespace(self) -> str | None:
        return self._request.namespace

    async def client[U: Resource](self, resource: type[U]) -> AsyncAPIClient[U]:
        return await self._request.client(resource)

    async def get[U: Resource](
        self, resource: type[U], name: str, *, namespace: str | None = None
    ) -> U:
        """Read live, defaulting to the primary namespace."""
        client = await self.client(resource)
        return await client.get(name, namespace=namespace)

    def cached[U: Resource](self, resource: type[U]) -> CachedResources[U]:
        return self._request.cached(resource)

    async def ensure[U: Resource](self, desired: U) -> U:
        return await self._request.ensure(desired)

    def set_status(self, **changes: object) -> None:
        self._request.set_status(**changes)

    def condition(
        self, name: str, status: bool | Literal["Unknown"], *, reason: str, message: str = ""
    ) -> None:
        if name in self._request._report.reserved:
            raise ValueError(f"Condition {name!r} is managed by the controller")
        self._request.condition(name, status, reason=reason, message=message)

    def event(
        self, reason: str, message: str, *, type: Literal["Normal", "Warning"] = "Normal"
    ) -> None:
        """Queue a bounded, best-effort Event, flushed after status persistence."""
        if not reason or type not in ("Normal", "Warning"):
            raise ValueError("Events need a reason and a Normal or Warning type")
        report = self._request._report
        if len(report.events) < 100:
            report.events.append((reason, message, type, report.action or "Reconcile"))

event(reason, message, *, type='Normal')

Queue a bounded, best-effort Event, flushed after status persistence.

Source code in cloudcoil/controller/_context.py
56
57
58
59
60
61
62
63
64
def event(
    self, reason: str, message: str, *, type: Literal["Normal", "Warning"] = "Normal"
) -> None:
    """Queue a bounded, best-effort Event, flushed after status persistence."""
    if not reason or type not in ("Normal", "Warning"):
        raise ValueError("Events need a reason and a Normal or Warning type")
    report = self._request._report
    if len(report.events) < 100:
        report.events.append((reason, message, type, report.action or "Reconcile"))

get(resource, name, *, namespace=None) async

Read live, defaulting to the primary namespace.

Source code in cloudcoil/controller/_context.py
33
34
35
36
37
38
async def get[U: Resource](
    self, resource: type[U], name: str, *, namespace: str | None = None
) -> U:
    """Read live, defaulting to the primary namespace."""
    client = await self.client(resource)
    return await client.get(name, namespace=namespace)

StageScope

One automatically executed stage, containing a handler or first-match cases.

Usually created through Controller.stage(). A decorator returns the original function, preserving its signature; a named scope can register case handlers.

Source code in cloudcoil/controller/_registry.py
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
class StageScope[T: Resource]:
    """One automatically executed stage, containing a handler or first-match cases.

    Usually created through Controller.stage(). A decorator returns the original
    function, preserving its signature; a named scope can register case handlers.
    """

    def __init__(
        self,
        registry: "Registry[T]",
        name: str | None,
        condition: str,
        depends: object | tuple[object, ...] | None,
    ) -> None:
        if not re.fullmatch(r"[A-Za-z][A-Za-z0-9_]{0,127}", condition) or condition == "Ready":
            raise ValueError("Stage condition must be an identifier other than Ready")
        self.registry = registry
        self.name = name or condition
        self.condition = condition
        self.dependencies = _refs(depends)
        self.handler: _Handler | None = None
        self.identity: object = self
        self._branches = _Branches(self._check_cases)

    def __call__[F: Callable[..., Any]](self, function: F) -> F:
        self.registry.check()
        if self.handler is not None or self._branches.populated:
            raise ValueError("A stage has one handler or cases, not both")
        self.handler = _Handler.build(function)
        self.identity = function
        return function

    def _check_cases(self) -> None:
        self.registry.check()
        if self.handler is not None:
            raise ValueError("A stage has one handler or cases, not both")

    def case[F: Callable[..., Any]](
        self, *, when: Callable[..., bool], after: object | tuple[object, ...] | None = None
    ) -> Callable[[F], F]:
        return self._branches.case(when=when, after=after)

    def otherwise[F: Callable[..., Any]](self) -> Callable[[F], F]:
        return self._branches.otherwise()

    def validate(self) -> None:
        if self.handler is None:
            self._branches.validate()

ResourceKey dataclass

Identity within a controller's primary resource kind; None is cluster scope.

Source code in cloudcoil/controller/_types.py
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
@dataclass(frozen=True)
class ResourceKey:
    """Identity within a controller's primary resource kind; None is cluster scope."""

    name: str
    namespace: str | None = None

    def __post_init__(self) -> None:
        if not self.name:
            raise ValueError("A resource key needs a name")

    @classmethod
    def from_resource(cls, resource: Resource) -> "ResourceKey":
        if not resource.name:
            raise ValueError("A resource key needs metadata.name")
        return cls(resource.name, resource.namespace)

Result dataclass

Successful reconciliation; optionally persist a resource and schedule another pass.

resource is a modified copy of this request's primary snapshot. The controller patches its differences before scheduling requeue_after. None performs no write.

Source code in cloudcoil/controller/_types.py
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
@dataclass(frozen=True)
class Result:
    """Successful reconciliation; optionally persist a resource and schedule another pass.

    resource is a modified copy of this request's primary snapshot. The controller
    patches its differences before scheduling requeue_after. None performs no write.
    """

    requeue_after: float | None = None
    resource: Resource | None = None

    def __post_init__(self) -> None:
        if self.requeue_after is not None and (
            not math.isfinite(self.requeue_after) or self.requeue_after < 0
        ):
            raise ValueError("requeue_after must be finite and nonnegative")

Wait

Bases: Exception

Expected pending work; raise to stop a pass without increasing backoff.

The low-level returned-Wait interface and requeue_after spelling remain usable.

Source code in cloudcoil/controller/_types.py
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
class Wait(Exception):
    """Expected pending work; raise to stop a pass without increasing backoff.

    The low-level returned-Wait interface and requeue_after spelling remain usable.
    """

    def __init__(
        self,
        reason: str,
        message: str = "",
        *,
        after: float | None = None,
        requeue_after: float | None = None,
    ) -> None:
        if after is not None and requeue_after is not None:
            raise ValueError("Use after or requeue_after, not both")
        delay = after if after is not None else requeue_after if requeue_after is not None else 30
        if not reason:
            raise ValueError("Wait needs a reason")
        if isinstance(delay, bool) or not math.isfinite(delay) or delay <= 0:
            raise ValueError("Wait delay must be finite and positive")
        self.reason = reason
        self.message = message
        self.requeue_after = delay
        super().__init__(message or reason)

TerminalError

Bases: Exception

Do not retry this failure; a later event or resync can still reconcile the key.

Source code in cloudcoil/controller/_types.py
233
234
class TerminalError(Exception):
    """Do not retry this failure; a later event or resync can still reconcile the key."""

ReconcileStatus

Bases: BaseModel

Optional base for CR status models used by staged reconcilers.

Source code in cloudcoil/controller/_status.py
15
16
17
18
19
20
21
class ReconcileStatus(BaseModel):
    """Optional base for CR status models used by staged reconcilers."""

    conditions: Annotated[list[Condition], ListType("map", keys=("type",))] = Field(
        default_factory=list
    )
    observed_generation: int | None = Field(default=None, alias="observedGeneration")

EventRecorder

Emit diagnostics, suppressing repeated reasons per resource UID for interval.

Memory is bounded by max_keys and traffic by a 20-event burst / 5 events per second token bucket per recorder. Changing messages do not defeat suppression. Delivery has a timeout and never raises an API/transport error to a reconciler. No background tasks or durable/exactly-once delivery guarantees.

Source code in cloudcoil/controller/_events.py
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
class EventRecorder:
    """Emit diagnostics, suppressing repeated reasons per resource UID for interval.

    Memory is bounded by max_keys and traffic by a 20-event burst / 5 events per
    second token bucket per recorder. Changing messages do not defeat suppression.
    Delivery has a timeout and never raises an API/transport error to a reconciler.
    No background tasks or durable/exactly-once delivery guarantees.
    """

    def __init__(
        self,
        reporting_controller: str = "cloudcoil",
        *,
        interval: float = 60,
        max_keys: int = 1024,
        timeout: float = 2,
        namespace: str | None = None,
    ) -> None:
        if not reporting_controller or len(reporting_controller) > 128:
            raise ValueError("reporting_controller must contain 1-128 characters")
        if not all(math.isfinite(v) and v > 0 for v in (interval, timeout)):
            raise ValueError("Event interval and timeout must be finite and positive")
        if isinstance(max_keys, bool) or not isinstance(max_keys, int) or max_keys < 1:
            raise ValueError("max_keys must be a positive integer")
        self.reporting_controller = reporting_controller
        self.namespace = namespace
        self.interval = interval
        self.max_keys = max_keys
        self.timeout = timeout
        self._instance = str(uuid4())
        self._recent: OrderedDict[tuple[str, ...], float] = OrderedDict()
        self._clock = time.monotonic
        self._tokens = 20.0
        self._last_token = self._clock()

    async def emit(
        self,
        resource: Resource,
        reason: str,
        message: str,
        *,
        type: Literal["Normal", "Warning"] = "Normal",
        action: str | None = None,
        config: Config | None = None,
    ) -> bool:
        """Return whether delivered; False means suppressed, disabled identity, or failed.

        Programmer errors (invalid reason/type/action) raise. Cancellation propagates.
        Cluster-scoped objects use the recorder namespace or the Config namespace.
        """
        if not re.fullmatch(r"[A-Za-z][A-Za-z0-9_]*", reason) or len(reason) > 128:
            raise ValueError("Event reason must be a nonempty identifier of at most 128 characters")
        if type not in ("Normal", "Warning"):
            raise ValueError("Event type must be Normal or Warning")
        action = reason if action is None else action
        if not action or len(action) > 128:
            raise ValueError("Event action must contain 1-128 characters")
        if not resource.name or not resource.metadata or not resource.metadata.uid:
            return False
        config = config or context.active_config
        namespace = resource.namespace or self.namespace or config.namespace
        key = (config.server or "", namespace, resource.metadata.uid, type, reason, action)
        now = self._clock()
        previous = self._recent.get(key)
        # Compare deadlines: subtraction can round an elapsed interval below its boundary.
        if previous is not None and now < previous + self.interval:
            return False
        self._tokens = min(20.0, self._tokens + (now - self._last_token) * 5)
        self._last_token = now
        if self._tokens < 1:
            return False
        self._tokens -= 1
        # Reserve before awaiting: concurrent calls for the same key coalesce too.
        self._recent[key] = now
        self._recent.move_to_end(key)
        while len(self._recent) > self.max_keys:
            self._recent.popitem(last=False)
        regarding = {
            "apiVersion": resource.api_version,
            "kind": resource.kind,
            "name": resource.name,
            "uid": resource.metadata.uid,
        }
        if resource.namespace:
            regarding["namespace"] = resource.namespace
        body = {
            "apiVersion": "events.k8s.io/v1",
            "kind": "Event",
            "metadata": {"name": f"cloudcoil-{uuid4().hex}", "namespace": namespace},
            "eventTime": datetime.now(timezone.utc).isoformat(timespec="microseconds"),
            "reportingController": self.reporting_controller,
            "reportingInstance": self._instance,
            "regarding": regarding,
            "reason": reason,
            "action": action,
            "note": message[:1024],
            "type": type,
        }
        try:
            async with asyncio.timeout(self.timeout):
                response = await config.async_client.post(
                    f"/apis/events.k8s.io/v1/namespaces/{quote(namespace, safe='')}/events",
                    json=body,
                )
                raise_for_status(response)
            return True
        except Exception:
            # Keep the reservation on failure to bound repeated denied/API requests.
            logger.warning(
                "Could not record Kubernetes Event %s for %s", reason, resource.name, exc_info=True
            )
            return False

emit(resource, reason, message, *, type='Normal', action=None, config=None) async

Return whether delivered; False means suppressed, disabled identity, or failed.

Programmer errors (invalid reason/type/action) raise. Cancellation propagates. Cluster-scoped objects use the recorder namespace or the Config namespace.

Source code in cloudcoil/controller/_events.py
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
async def emit(
    self,
    resource: Resource,
    reason: str,
    message: str,
    *,
    type: Literal["Normal", "Warning"] = "Normal",
    action: str | None = None,
    config: Config | None = None,
) -> bool:
    """Return whether delivered; False means suppressed, disabled identity, or failed.

    Programmer errors (invalid reason/type/action) raise. Cancellation propagates.
    Cluster-scoped objects use the recorder namespace or the Config namespace.
    """
    if not re.fullmatch(r"[A-Za-z][A-Za-z0-9_]*", reason) or len(reason) > 128:
        raise ValueError("Event reason must be a nonempty identifier of at most 128 characters")
    if type not in ("Normal", "Warning"):
        raise ValueError("Event type must be Normal or Warning")
    action = reason if action is None else action
    if not action or len(action) > 128:
        raise ValueError("Event action must contain 1-128 characters")
    if not resource.name or not resource.metadata or not resource.metadata.uid:
        return False
    config = config or context.active_config
    namespace = resource.namespace or self.namespace or config.namespace
    key = (config.server or "", namespace, resource.metadata.uid, type, reason, action)
    now = self._clock()
    previous = self._recent.get(key)
    # Compare deadlines: subtraction can round an elapsed interval below its boundary.
    if previous is not None and now < previous + self.interval:
        return False
    self._tokens = min(20.0, self._tokens + (now - self._last_token) * 5)
    self._last_token = now
    if self._tokens < 1:
        return False
    self._tokens -= 1
    # Reserve before awaiting: concurrent calls for the same key coalesce too.
    self._recent[key] = now
    self._recent.move_to_end(key)
    while len(self._recent) > self.max_keys:
        self._recent.popitem(last=False)
    regarding = {
        "apiVersion": resource.api_version,
        "kind": resource.kind,
        "name": resource.name,
        "uid": resource.metadata.uid,
    }
    if resource.namespace:
        regarding["namespace"] = resource.namespace
    body = {
        "apiVersion": "events.k8s.io/v1",
        "kind": "Event",
        "metadata": {"name": f"cloudcoil-{uuid4().hex}", "namespace": namespace},
        "eventTime": datetime.now(timezone.utc).isoformat(timespec="microseconds"),
        "reportingController": self.reporting_controller,
        "reportingInstance": self._instance,
        "regarding": regarding,
        "reason": reason,
        "action": action,
        "note": message[:1024],
        "type": type,
    }
    try:
        async with asyncio.timeout(self.timeout):
            response = await config.async_client.post(
                f"/apis/events.k8s.io/v1/namespaces/{quote(namespace, safe='')}/events",
                json=body,
            )
            raise_for_status(response)
        return True
    except Exception:
        # Keep the reservation on failure to bound repeated denied/API requests.
        logger.warning(
            "Could not record Kubernetes Event %s for %s", reason, resource.name, exc_info=True
        )
        return False

get_condition(resource, condition)

Read a standard metav1 condition by type, returning an independent copy.

Source code in cloudcoil/controller/_status.py
56
57
58
59
60
61
62
def get_condition(resource: Resource, condition: str) -> Condition | None:
    """Read a standard metav1 condition by type, returning an independent copy."""
    status = getattr(resource, "status", None)
    for value in getattr(status, "conditions", None) or []:
        if value.type == condition:
            return Condition.model_validate(value.model_dump(by_alias=True)).model_copy(deep=True)
    return None

set_condition(resource, condition, status, *, reason, message='')

Upsert a standard condition without churning lastTransitionTime.

Only a status change updates the timestamp; reason/message/generation changes preserve it. Other condition types and status fields survive. No API I/O.

Source code in cloudcoil/controller/_status.py
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
def set_condition[T: Resource](
    resource: T,
    condition: str,
    status: bool | Literal["True", "False", "Unknown"],
    *,
    reason: str,
    message: str = "",
) -> T:
    """Upsert a standard condition without churning lastTransitionTime.

    Only a status change updates the timestamp; reason/message/generation changes
    preserve it. Other condition types and status fields survive. No API I/O.
    """
    if not condition or not reason:
        raise ValueError("Condition type and reason must not be empty")
    value = str(status) if isinstance(status, bool) else status
    if value not in ("True", "False", "Unknown"):
        raise ValueError("Condition status must be True, False, or Unknown")
    model = _status_model(resource)
    if "conditions" not in model.model_fields:
        raise TypeError("Status conditions must be declared in the resource schema")
    previous = get_condition(resource, condition)
    updated = Condition(
        type=condition,
        status=value,
        reason=reason,
        message=message,
        observed_generation=resource.metadata.generation if resource.metadata else None,
        last_transition_time=(
            previous.last_transition_time
            if previous is not None and previous.status == value
            else Time(datetime.now(timezone.utc))
        ),
    )
    current = getattr(resource, "status", None)
    conditions = list(getattr(current, "conditions", None) or [])
    index = next(
        (i for i, item in enumerate(conditions) if item.type == condition), len(conditions)
    )
    # Remove malformed duplicate entries for this type, preserving all other types.
    conditions = [item for item in conditions if item.type != condition]
    conditions.insert(index, updated)
    field = model.model_fields["conditions"]
    values = [item.model_dump(by_alias=True) for item in conditions]
    return update_status(resource, conditions=TypeAdapter(field.annotation).validate_python(values))

update_status(resource, **changes)

Edit supplied status fields, preserving other fields; return the resource.

Accepts Python field names or wire aliases. Creates an absent status through its schema (required fields must be provided), validates updates, and rejects typos. This is a local edit: return the resource from an ordinary reconciler to save it.

Source code in cloudcoil/controller/_status.py
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
def update_status[T: Resource](resource: T, **changes: Any) -> T:
    """Edit supplied status fields, preserving other fields; return the resource.

    Accepts Python field names or wire aliases. Creates an absent status through its
    schema (required fields must be provided), validates updates, and rejects typos.
    This is a local edit: return the resource from an ordinary reconciler to save it.
    """
    model = _status_model(resource)
    current = getattr(resource, "status", None)
    values = current.model_dump(mode="python", by_alias=True) if current is not None else {}
    for name, value in changes.items():
        field_name = next(
            (key for key, field in model.model_fields.items() if name in (key, field.alias)), None
        )
        if field_name is None:
            raise ValueError(f"Unknown status field {name!r} on {model.__name__}")
        field = model.model_fields[field_name]
        values[field.alias or field_name] = value
    # Validate before assignment so a failure leaves the original status untouched.
    resource.status = model.model_validate(values)  # type: ignore[attr-defined]
    return resource

Applications

Shared operator manifests, installation, and process entry point.

Application

Describe, install, and run controllers and admission policies.

Construction and manifest generation are offline. Config is created lazily; an explicitly passed Config remains owned by its caller. Installation uses the caller's privileges; generated runtime RBAC never grants CRD/RBAC setup privileges implicitly. Declare extra RBAC rules for application API calls.

Source code in cloudcoil/application/_application.py
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
class Application:
    """Describe, install, and run controllers and admission policies.

    Construction and manifest generation are offline. Config is created lazily;
    an explicitly passed Config remains owned by its caller. Installation uses
    the caller's privileges; generated runtime RBAC never grants CRD/RBAC setup
    privileges implicitly. Declare extra RBAC rules for application API calls.
    """

    def __init__(
        self,
        name: str,
        *controllers: Controller[Any],
        resources: Sequence[type[Resource] | CRD] = (),
        namespace: str | None = None,
        config: Config | None = None,
        cache: Cache | None = None,
        rules: Sequence[RBACRule] = (),
        admission: AdmissionWebhook | None = None,
        webhook: WebhookServer | None = None,
        leader_election: LeaderElection | bool | None = None,
        health: HealthServer | None = None,
    ) -> None:
        namespace = namespace or (
            config.namespace if config else os.environ.get("CLOUDCOIL_NAMESPACE", "default")
        )
        self.name = name
        self.namespace = namespace
        self.controllers: tuple[Controller[Any], ...] = ()
        self._frozen = False
        self._resources = tuple(resources)
        self._lifespans = Lifespans(self._check_registration)
        self.rules = tuple(rules)
        self.webhook = webhook
        leader_election = (
            LeaderElection(name) if leader_election is True else leader_election or None
        )
        self.leader_election = leader_election
        self.health = health
        self.config = config
        self.cache = cache
        if config is not None and cache is not None:
            raise ValueError("Configure caching on Config or Application, not both")
        self.admission = admission
        self._admission_registry = AdmissionRegistry(
            admission if admission is not None else AdmissionWebhook(), self._check_registration
        )
        if admission is not None and admission._config not in (None, config):
            raise ValueError("AdmissionWebhook must share the operator Config")
        if config is not None and config.namespace != namespace:
            raise ValueError("Config.namespace must match Application.namespace")
        self.crds: tuple[CRD, ...] = ()
        self._models: tuple[type[Resource], ...] = ()
        self._has_admission = False
        if leader_election and leader_election.config not in (None, config):
            raise ValueError("Application leader election must share the operator Config")
        for controller in controllers:
            self.include(controller)
        self.manager: Manager | None = None
        self._used = False
        self._running = False
        self._serving = False
        self._failure: BaseException | None = None
        self._ready = asyncio.Event()
        self._finished = asyncio.Event()

    def _check_registration(self) -> None:
        if self._frozen:
            raise RuntimeError("Register application components before running")

    def include(self, controller: Controller[Any]) -> Self:
        """Explicitly include a reusable controller group."""
        self._check_registration()
        if any(item is controller for item in self.controllers):
            raise ValueError("A controller cannot be registered twice")
        if controller.config is not None and controller.config is not self.config:
            raise ValueError("Application controllers must share the operator Config")
        self.controllers = (*self.controllers, controller)
        return self

    def controller[T: Resource](self, resource: type[T], **options: Any) -> Controller[T]:
        """Create and include a controller registry."""
        controller = Controller(resource, **options)
        self.include(controller)
        return controller

    def validate[T: Resource](
        self, model: type[T], **options: Any
    ) -> Callable[[Validator[T]], Validator[T]]:
        return self._admission_registry.validate(model, **options)

    def mutate[T: Resource](
        self, model: type[T], **options: Any
    ) -> Callable[[Mutator[T]], Mutator[T]]:
        return self._admission_registry.mutate(model, **options)

    def lifespan[F: Callable[..., AsyncIterator[None]]](
        self, *, scope: Literal["process", "leader"] = "process"
    ) -> Callable[[F], F]:
        """Pair process or elected-leader startup and shutdown around handlers."""
        return self._lifespans.register(scope=scope)

    def _validate(self, *, freeze: bool = False) -> None:
        definitions: dict[type[Resource], CRD] = {}
        for item in self._resources:
            definition = item if isinstance(item, CRD) else CRD(item)
            if definition.resource in definitions:
                raise ValueError("A resource cannot be registered twice")
            definitions[definition.resource] = definition
        for controller in self.controllers:
            controller._validate()
            if controller.resource not in definitions and _resource_options(controller.resource):
                definitions[controller.resource] = CRD(controller.resource)
        self.crds = tuple(definitions.values())
        self._models = tuple(crd.resource for crd in self.crds if _methods(crd.resource))
        admission = self._admission()
        self._has_admission = bool(admission._routes)
        if self._has_admission and self.webhook is None:
            raise ValueError("Admission routes require webhook=WebhookServer(...)")
        if self.webhook is not None and not self._has_admission:
            raise ValueError("A webhook server needs admission routes")
        if not self.controllers and self.webhook is None:
            raise ValueError("An application needs controllers or webhooks")
        if "leader" in self._lifespans.handlers and (
            self.leader_election is None or not self.controllers
        ):
            raise ValueError("Leader lifespan requires leader election and controllers")
        if freeze:
            for controller in self.controllers:
                controller._validate(freeze=True)
            self._frozen = True

    def _admission(self, config: Config | None = None) -> AdmissionWebhook:
        admission = AdmissionWebhook(
            config=config,
            **({"max_body_bytes": self.admission._max_body_bytes} if self.admission else {}),
        )
        registries = [self._admission_registry, *(c._admission_registry for c in self.controllers)]
        for registry in registries:
            if set(admission._routes) & set(registry.webhook._routes):
                raise ValueError("Admission paths must be unique across included groups")
            admission._routes.update(registry.webhook._routes)
        admission._register_models(self._models, require_config=config is not None)
        if set(admission._routes) & {"/readyz", "/controllers/readyz", "/metrics"}:
            raise ValueError("Admission paths conflict with operator health/metrics endpoints")
        return admission

    def manifests(
        self,
        *,
        image: str | None = None,
        command: Sequence[str] | None = None,
        replicas: int = 1,
        include_webhooks: bool = True,
    ) -> list[dict[str, Any]]:
        """Return fresh CRD/RBAC/Service/Deployment/admission manifests, offline.

        image enables a Deployment; command replaces its container entry point.
        TLS Secrets and namespaces must already exist. include_webhooks=False
        exports the CRD/RBAC foundation without admission registration or hosting.
        """
        self._validate()
        if image and self.webhook and not include_webhooks:
            raise ValueError(
                "A Deployment for this operator requires its webhook TLS configuration"
            )
        return build_manifests(
            name=self.name,
            namespace=self.namespace,
            controllers=self.controllers,
            crds=self.crds,
            rules=self.rules,
            leader_election=self.leader_election,
            admission=self._admission() if include_webhooks and self._has_admission else None,
            webhook=self.webhook if include_webhooks else None,
            image=image,
            command=command,
            replicas=replicas,
        )

    def to_yaml(self, **options: Any) -> str:
        """Serialize manifests for review, kubectl, or GitOps."""
        return yaml.safe_dump_all(self.manifests(**options), sort_keys=False)

    @asynccontextmanager
    async def _configuration(self):
        owned = self.config is None
        config = (
            self.config
            if self.config is not None
            else Config(
                namespace=self.namespace, cache=self.cache if self.cache is not None else False
            )
        )
        try:
            yield config
        finally:
            if owned:
                try:
                    await config.async_client.aclose()
                finally:
                    await asyncio.to_thread(config.client.close)

    async def install(
        self,
        *,
        image: str | None = None,
        command: Sequence[str] | None = None,
        replicas: int = 1,
        include_webhooks: bool = True,
        timeout: float = 120,
        force: bool = False,
    ) -> None:
        """Apply desired objects, establish CRDs, and enable webhooks after rollout.

        Existing objects are updated with server-side apply; ownership conflicts
        fail unless force=True. A failure leaves already applied objects in place
        for inspection and retry. No resources are deleted or rolled back.
        """
        if not math.isfinite(timeout) or timeout <= 0:
            raise ValueError("Installation timeout must be finite and positive")
        if include_webhooks and self.webhook and image is None:
            raise ValueError(
                "Installing webhooks requires image=... to wait for a serving Deployment; "
                "use include_webhooks=False to install only CRDs and RBAC"
            )
        manifests = self.manifests(
            image=image, command=command, replicas=replicas, include_webhooks=include_webhooks
        )
        async with self._configuration() as config:
            await install(
                config,
                manifests,
                field_manager=f"cloudcoil-{self.name}",
                timeout=timeout,
                force=force,
                wait_deployment=image is not None,
            )

    @property
    def ready(self) -> bool:
        """Admission readiness on every replica; controller readiness is separate."""
        return (
            self.healthy
            and self._ready.is_set()
            and (self.webhook is not None or (self.manager is not None and self.manager.ready))
        )

    @property
    def healthy(self) -> bool:
        return (
            self._running
            and self._failure is None
            and (self.webhook is None or self._serving)
            and (self.manager is None or self.manager.healthy)
        )

    async def wait_ready(self, timeout: float = 30) -> None:
        async with asyncio.timeout(timeout):
            while not self.ready:
                if self._failure is not None:
                    raise self._failure
                if self._finished.is_set():
                    raise RuntimeError("Application stopped before becoming ready")
                await asyncio.sleep(0.01)

    def _application(self, admission: AdmissionWebhook):
        async def app(scope: dict[str, Any], receive: Any, send: Any) -> None:
            if scope["type"] == "http" and scope.get("path") in (
                "/readyz",
                "/healthz",
                "/controllers/readyz",
                "/metrics",
            ):
                path = scope["path"]
                if path == "/metrics":
                    code, body = 200, self.manager.metrics() if self.manager else ""
                else:
                    good = self.ready if path == "/readyz" else self.healthy
                    if path == "/controllers/readyz":
                        good = self.manager is not None and self.manager.ready
                    code, body = (200, "ok\n") if good else (503, "not ready\n")
                await send(
                    {
                        "type": "http.response.start",
                        "status": code,
                        "headers": [(b"content-type", b"text/plain; charset=utf-8")],
                    }
                )
                await send({"type": "http.response.body", "body": body.encode()})
                return
            await admission(scope, receive, send)

        return app

    async def run(self, *, stop: asyncio.Event | None = None) -> None:
        """Serve admission on every replica and run the manager until stopped.

        Installation is an explicit earlier step. Workers start after discovery;
        manager leader election does not gate webhook serving. Embedded use does
        not replace signal handlers; main() supplies SIGINT/SIGTERM shutdown.
        """
        if self._used:
            raise RuntimeError("Application instances can only run once")
        self._validate(freeze=True)
        self._used = True
        stop = stop if stop is not None else asyncio.Event()
        component_stop = asyncio.Event()
        server: _HTTPS | None = None
        tasks: list[asyncio.Task[Any]] = []

        async def shutdown(exc_type: Any, exception: BaseException | None, traceback: Any) -> bool:
            self._ready.clear()
            component_stop.set()
            errors: list[BaseException] = []
            try:
                if server:
                    await server.close()
            except Exception as error:
                errors.append(error)
            finally:
                for task in tasks:
                    if not task.done():
                        task.cancel()
                outcomes = await asyncio.gather(*tasks, return_exceptions=True)
                for outcome in outcomes:
                    if (
                        isinstance(outcome, BaseException)
                        and not isinstance(outcome, asyncio.CancelledError)
                        and outcome is not exception
                        and all(outcome is not error for error in errors)
                    ):
                        errors.append(outcome)
                self._serving = False
                self._running = False
            if errors:
                if exception is not None and not isinstance(exception, asyncio.CancelledError):
                    errors.insert(0, exception)
                if len(errors) == 1:
                    raise errors[0]
                raise BaseExceptionGroup("Application failed while stopping components", errors)
            return False

        async def run_components(config: Config) -> None:
            nonlocal server
            async with AsyncExitStack() as stack:
                await stack.enter_async_context(config)
                await stack.enter_async_context(
                    self._lifespans.enter("process", self.leader_election)
                )
                stack.push_async_exit(shutdown)
                self._running = True
                admission = self._admission(config)
                if self.controllers:
                    self.manager = Manager(
                        *self.controllers,
                        config=config,
                        leader_election=self.leader_election,
                        health=self.health,
                        leader_lifespan=lambda: self._lifespans.enter(
                            "leader", self.leader_election
                        ),
                    )
                if self.webhook:
                    server = _HTTPS(self._application(admission), self.webhook)
                    await server.start()
                    self._serving = True
                    assert server.task is not None
                    tasks.append(server.task)
                if self.manager:
                    tasks.append(asyncio.create_task(self.manager.run(stop=component_stop)))
                    await asyncio.sleep(0)
                self._ready.set()
                done, _ = await asyncio.wait(tasks, return_when=asyncio.FIRST_COMPLETED)
                for task in done:
                    await task
                if not stop.is_set():
                    raise RuntimeError("An operator component stopped unexpectedly")

        runtime: asyncio.Task[None] | None = None
        stopped: asyncio.Task[bool] | None = None
        try:
            if stop.is_set():
                return
            async with self._configuration() as config:
                if config.namespace != self.namespace:
                    raise ValueError("Config.namespace must match Application.namespace")
                runtime = asyncio.create_task(run_components(config))
                stopped = asyncio.create_task(stop.wait())
                try:
                    done, _ = await asyncio.wait(
                        (runtime, stopped), return_when=asyncio.FIRST_COMPLETED
                    )
                    if runtime in done:
                        await runtime
                finally:
                    # Covers stop during discovery, TLS startup, and informer sync.
                    runtime.cancel()
                    stopped.cancel()
                    outcomes = await asyncio.gather(runtime, stopped, return_exceptions=True)
                    error = outcomes[0]
                    if isinstance(error, BaseException) and not isinstance(
                        error, asyncio.CancelledError
                    ):
                        raise error
        except BaseException as exc:
            self._failure = exc
            raise
        finally:
            self._finished.set()

    def main(self, argv: Sequence[str] | None = None) -> None:
        """Shared ``manifests``, ``install``, and ``run`` command-line entry point."""
        parser = argparse.ArgumentParser(description=f"Manage the {self.name} operator")
        commands = parser.add_subparsers(dest="command_name", required=True)
        for name in ("manifests", "install"):
            command = commands.add_parser(name)
            command.add_argument("--image")
            command.add_argument(
                "--command", type=shlex.split, help="Quoted container command before run (no shell)"
            )
            command.add_argument("--replicas", type=int, default=1)
            command.add_argument("--without-webhooks", action="store_true")
            command.add_argument(
                "--ca-file",
                default=os.environ.get("CLOUDCOIL_WEBHOOK_CA_FILE"),
                help="Public PEM CA for webhook registration (or CLOUDCOIL_WEBHOOK_CA_FILE)",
            )
            if name == "install":
                command.add_argument("--timeout", type=float, default=120)
                command.add_argument("--force", action="store_true")
        commands.add_parser("run")
        args = parser.parse_args(argv)
        if args.command_name == "run":
            asyncio.run(self._main_run())
        else:
            if args.ca_file:
                if self.webhook is None:
                    parser.error("--ca-file requires a webhook server")
                self.webhook = replace(self.webhook, ca_bundle=Path(args.ca_file).read_bytes())
            options = dict(
                image=args.image,
                command=args.command,
                replicas=args.replicas,
                include_webhooks=not args.without_webhooks,
            )
            if args.command_name == "manifests":
                print(self.to_yaml(**options), end="")
            else:
                asyncio.run(self.install(**options, timeout=args.timeout, force=args.force))

    async def _main_run(self) -> None:
        stop = asyncio.Event()
        loop = asyncio.get_running_loop()
        previous: dict[signal.Signals, Any] = {}
        try:
            for sig in (signal.SIGINT, signal.SIGTERM):
                previous[sig] = signal.getsignal(sig)
                loop.add_signal_handler(sig, stop.set)
            await self.run(stop=stop)
        finally:
            for sig, handler in previous.items():
                loop.remove_signal_handler(sig)
                signal.signal(sig, handler)

ready property

Admission readiness on every replica; controller readiness is separate.

controller(resource, **options)

Create and include a controller registry.

Source code in cloudcoil/application/_application.py
113
114
115
116
117
def controller[T: Resource](self, resource: type[T], **options: Any) -> Controller[T]:
    """Create and include a controller registry."""
    controller = Controller(resource, **options)
    self.include(controller)
    return controller

include(controller)

Explicitly include a reusable controller group.

Source code in cloudcoil/application/_application.py
103
104
105
106
107
108
109
110
111
def include(self, controller: Controller[Any]) -> Self:
    """Explicitly include a reusable controller group."""
    self._check_registration()
    if any(item is controller for item in self.controllers):
        raise ValueError("A controller cannot be registered twice")
    if controller.config is not None and controller.config is not self.config:
        raise ValueError("Application controllers must share the operator Config")
    self.controllers = (*self.controllers, controller)
    return self

install(*, image=None, command=None, replicas=1, include_webhooks=True, timeout=120, force=False) async

Apply desired objects, establish CRDs, and enable webhooks after rollout.

Existing objects are updated with server-side apply; ownership conflicts fail unless force=True. A failure leaves already applied objects in place for inspection and retry. No resources are deleted or rolled back.

Source code in cloudcoil/application/_application.py
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
async def install(
    self,
    *,
    image: str | None = None,
    command: Sequence[str] | None = None,
    replicas: int = 1,
    include_webhooks: bool = True,
    timeout: float = 120,
    force: bool = False,
) -> None:
    """Apply desired objects, establish CRDs, and enable webhooks after rollout.

    Existing objects are updated with server-side apply; ownership conflicts
    fail unless force=True. A failure leaves already applied objects in place
    for inspection and retry. No resources are deleted or rolled back.
    """
    if not math.isfinite(timeout) or timeout <= 0:
        raise ValueError("Installation timeout must be finite and positive")
    if include_webhooks and self.webhook and image is None:
        raise ValueError(
            "Installing webhooks requires image=... to wait for a serving Deployment; "
            "use include_webhooks=False to install only CRDs and RBAC"
        )
    manifests = self.manifests(
        image=image, command=command, replicas=replicas, include_webhooks=include_webhooks
    )
    async with self._configuration() as config:
        await install(
            config,
            manifests,
            field_manager=f"cloudcoil-{self.name}",
            timeout=timeout,
            force=force,
            wait_deployment=image is not None,
        )

lifespan(*, scope='process')

Pair process or elected-leader startup and shutdown around handlers.

Source code in cloudcoil/application/_application.py
129
130
131
132
133
def lifespan[F: Callable[..., AsyncIterator[None]]](
    self, *, scope: Literal["process", "leader"] = "process"
) -> Callable[[F], F]:
    """Pair process or elected-leader startup and shutdown around handlers."""
    return self._lifespans.register(scope=scope)

main(argv=None)

Shared manifests, install, and run command-line entry point.

Source code in cloudcoil/application/_application.py
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
def main(self, argv: Sequence[str] | None = None) -> None:
    """Shared ``manifests``, ``install``, and ``run`` command-line entry point."""
    parser = argparse.ArgumentParser(description=f"Manage the {self.name} operator")
    commands = parser.add_subparsers(dest="command_name", required=True)
    for name in ("manifests", "install"):
        command = commands.add_parser(name)
        command.add_argument("--image")
        command.add_argument(
            "--command", type=shlex.split, help="Quoted container command before run (no shell)"
        )
        command.add_argument("--replicas", type=int, default=1)
        command.add_argument("--without-webhooks", action="store_true")
        command.add_argument(
            "--ca-file",
            default=os.environ.get("CLOUDCOIL_WEBHOOK_CA_FILE"),
            help="Public PEM CA for webhook registration (or CLOUDCOIL_WEBHOOK_CA_FILE)",
        )
        if name == "install":
            command.add_argument("--timeout", type=float, default=120)
            command.add_argument("--force", action="store_true")
    commands.add_parser("run")
    args = parser.parse_args(argv)
    if args.command_name == "run":
        asyncio.run(self._main_run())
    else:
        if args.ca_file:
            if self.webhook is None:
                parser.error("--ca-file requires a webhook server")
            self.webhook = replace(self.webhook, ca_bundle=Path(args.ca_file).read_bytes())
        options = dict(
            image=args.image,
            command=args.command,
            replicas=args.replicas,
            include_webhooks=not args.without_webhooks,
        )
        if args.command_name == "manifests":
            print(self.to_yaml(**options), end="")
        else:
            asyncio.run(self.install(**options, timeout=args.timeout, force=args.force))

manifests(*, image=None, command=None, replicas=1, include_webhooks=True)

Return fresh CRD/RBAC/Service/Deployment/admission manifests, offline.

image enables a Deployment; command replaces its container entry point. TLS Secrets and namespaces must already exist. include_webhooks=False exports the CRD/RBAC foundation without admission registration or hosting.

Source code in cloudcoil/application/_application.py
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
def manifests(
    self,
    *,
    image: str | None = None,
    command: Sequence[str] | None = None,
    replicas: int = 1,
    include_webhooks: bool = True,
) -> list[dict[str, Any]]:
    """Return fresh CRD/RBAC/Service/Deployment/admission manifests, offline.

    image enables a Deployment; command replaces its container entry point.
    TLS Secrets and namespaces must already exist. include_webhooks=False
    exports the CRD/RBAC foundation without admission registration or hosting.
    """
    self._validate()
    if image and self.webhook and not include_webhooks:
        raise ValueError(
            "A Deployment for this operator requires its webhook TLS configuration"
        )
    return build_manifests(
        name=self.name,
        namespace=self.namespace,
        controllers=self.controllers,
        crds=self.crds,
        rules=self.rules,
        leader_election=self.leader_election,
        admission=self._admission() if include_webhooks and self._has_admission else None,
        webhook=self.webhook if include_webhooks else None,
        image=image,
        command=command,
        replicas=replicas,
    )

run(*, stop=None) async

Serve admission on every replica and run the manager until stopped.

Installation is an explicit earlier step. Workers start after discovery; manager leader election does not gate webhook serving. Embedded use does not replace signal handlers; main() supplies SIGINT/SIGTERM shutdown.

Source code in cloudcoil/application/_application.py
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
async def run(self, *, stop: asyncio.Event | None = None) -> None:
    """Serve admission on every replica and run the manager until stopped.

    Installation is an explicit earlier step. Workers start after discovery;
    manager leader election does not gate webhook serving. Embedded use does
    not replace signal handlers; main() supplies SIGINT/SIGTERM shutdown.
    """
    if self._used:
        raise RuntimeError("Application instances can only run once")
    self._validate(freeze=True)
    self._used = True
    stop = stop if stop is not None else asyncio.Event()
    component_stop = asyncio.Event()
    server: _HTTPS | None = None
    tasks: list[asyncio.Task[Any]] = []

    async def shutdown(exc_type: Any, exception: BaseException | None, traceback: Any) -> bool:
        self._ready.clear()
        component_stop.set()
        errors: list[BaseException] = []
        try:
            if server:
                await server.close()
        except Exception as error:
            errors.append(error)
        finally:
            for task in tasks:
                if not task.done():
                    task.cancel()
            outcomes = await asyncio.gather(*tasks, return_exceptions=True)
            for outcome in outcomes:
                if (
                    isinstance(outcome, BaseException)
                    and not isinstance(outcome, asyncio.CancelledError)
                    and outcome is not exception
                    and all(outcome is not error for error in errors)
                ):
                    errors.append(outcome)
            self._serving = False
            self._running = False
        if errors:
            if exception is not None and not isinstance(exception, asyncio.CancelledError):
                errors.insert(0, exception)
            if len(errors) == 1:
                raise errors[0]
            raise BaseExceptionGroup("Application failed while stopping components", errors)
        return False

    async def run_components(config: Config) -> None:
        nonlocal server
        async with AsyncExitStack() as stack:
            await stack.enter_async_context(config)
            await stack.enter_async_context(
                self._lifespans.enter("process", self.leader_election)
            )
            stack.push_async_exit(shutdown)
            self._running = True
            admission = self._admission(config)
            if self.controllers:
                self.manager = Manager(
                    *self.controllers,
                    config=config,
                    leader_election=self.leader_election,
                    health=self.health,
                    leader_lifespan=lambda: self._lifespans.enter(
                        "leader", self.leader_election
                    ),
                )
            if self.webhook:
                server = _HTTPS(self._application(admission), self.webhook)
                await server.start()
                self._serving = True
                assert server.task is not None
                tasks.append(server.task)
            if self.manager:
                tasks.append(asyncio.create_task(self.manager.run(stop=component_stop)))
                await asyncio.sleep(0)
            self._ready.set()
            done, _ = await asyncio.wait(tasks, return_when=asyncio.FIRST_COMPLETED)
            for task in done:
                await task
            if not stop.is_set():
                raise RuntimeError("An operator component stopped unexpectedly")

    runtime: asyncio.Task[None] | None = None
    stopped: asyncio.Task[bool] | None = None
    try:
        if stop.is_set():
            return
        async with self._configuration() as config:
            if config.namespace != self.namespace:
                raise ValueError("Config.namespace must match Application.namespace")
            runtime = asyncio.create_task(run_components(config))
            stopped = asyncio.create_task(stop.wait())
            try:
                done, _ = await asyncio.wait(
                    (runtime, stopped), return_when=asyncio.FIRST_COMPLETED
                )
                if runtime in done:
                    await runtime
            finally:
                # Covers stop during discovery, TLS startup, and informer sync.
                runtime.cancel()
                stopped.cancel()
                outcomes = await asyncio.gather(runtime, stopped, return_exceptions=True)
                error = outcomes[0]
                if isinstance(error, BaseException) and not isinstance(
                    error, asyncio.CancelledError
                ):
                    raise error
    except BaseException as exc:
        self._failure = exc
        raise
    finally:
        self._finished.set()

to_yaml(**options)

Serialize manifests for review, kubectl, or GitOps.

Source code in cloudcoil/application/_application.py
213
214
215
def to_yaml(self, **options: Any) -> str:
    """Serialize manifests for review, kubectl, or GitOps."""
    return yaml.safe_dump_all(self.manifests(**options), sort_keys=False)

LifecycleEvent dataclass

A per-scope event whose type/error are updated before lifespan cleanup.

Read type inside the handler's finally block to distinguish normal shutdown, leadership loss, and failure. identity is populated for elected leaders.

Source code in cloudcoil/application/_lifecycle.py
22
23
24
25
26
27
28
29
30
31
32
33
@dataclass
class LifecycleEvent:
    """A per-scope event whose type/error are updated before lifespan cleanup.

    Read type inside the handler's finally block to distinguish normal shutdown,
    leadership loss, and failure. identity is populated for elected leaders.
    """

    type: LifecycleType
    scope: Literal["process", "leader"]
    identity: str | None = None
    error: BaseException | None = None

LifecycleType

Bases: StrEnum

Source code in cloudcoil/application/_lifecycle.py
14
15
16
17
18
19
class LifecycleType(StrEnum):
    STARTUP = "startup"
    LEADERSHIP_ACQUIRED = "leadership_acquired"
    SHUTDOWN = "shutdown"
    LEADERSHIP_LOST = "leadership_lost"
    FAILURE = "failure"

RBACRule dataclass

Additional access, with offline API metadata for generated resource classes.

Custom resources use their CRD declaration; generated models carry API metadata. Older models require plural and scope once. Namespaced access defaults to the operator's namespace; use all_namespaces=True explicitly for a cluster-wide grant. subresources selects those endpoints instead of the main resource. Reconcile and admission function bodies cannot be inspected to infer their additional access needs.

Source code in cloudcoil/application/_manifests.py
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
@dataclass(frozen=True)
class RBACRule:
    """Additional access, with offline API metadata for generated resource classes.

    Custom resources use their CRD declaration; generated models carry API metadata.
    Older models require ``plural`` and ``scope`` once. Namespaced access defaults to the operator's namespace;
    use ``all_namespaces=True`` explicitly for a cluster-wide grant. ``subresources``
    selects those endpoints instead of the main resource. Reconcile and admission
    function bodies cannot be inspected to infer their additional access needs.
    """

    resource: type[Resource]
    verbs: tuple[str, ...]
    plural: str | None = None
    scope: Scope | None = None
    namespace: str | None = None
    all_namespaces: bool = False
    subresources: tuple[str, ...] = ()
    resource_names: tuple[str, ...] = ()

    def __post_init__(self) -> None:
        if not isinstance(self.resource, type) or not issubclass(self.resource, Resource):
            raise TypeError("RBACRule.resource must be a Resource class")
        for field in ("verbs", "subresources", "resource_names"):
            values = getattr(self, field)
            if isinstance(values, str) or any(
                not isinstance(value, str) or not value for value in values
            ):
                raise ValueError(f"RBACRule.{field} must contain nonempty strings")
            object.__setattr__(self, field, tuple(sorted(set(values))))
        if not self.verbs:
            raise ValueError("RBACRule.verbs must not be empty")
        if self.plural is not None:
            _label(self.plural, "Resource plural")
        if self.scope not in (None, "Namespaced", "Cluster"):
            raise ValueError("RBACRule.scope must be Namespaced or Cluster")
        if self.namespace is not None:
            _label(self.namespace, "RBAC namespace")
        if self.namespace is not None and self.all_namespaces:
            raise ValueError("RBACRule.namespace and all_namespaces are mutually exclusive")
        if self.scope == "Cluster" and self.namespace is not None:
            raise ValueError("Cluster resources cannot have a namespaced RBACRule")
        if self.resource_names and set(self.verbs) & {"create", "deletecollection"}:
            raise ValueError("Kubernetes cannot restrict create/deletecollection by resource_names")
        for subresource in self.subresources:
            _label(subresource, "Subresource")

WebhookServer dataclass

TLS files are provided by the deployment, typically from a mounted Secret.

The certificate must cover <operator-name>.<namespace>.svc. ca_bundle contains public PEM certificates only; private keys never enter manifests.

Source code in cloudcoil/application/_server.py
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
@dataclass(frozen=True)
class WebhookServer:
    """TLS files are provided by the deployment, typically from a mounted Secret.

    The certificate must cover ``<operator-name>.<namespace>.svc``. ``ca_bundle``
    contains public PEM certificates only; private keys never enter manifests.
    """

    ca_bundle: bytes = b""
    tls_secret: str = "operator-tls"
    certfile: str = "/var/run/cloudcoil/tls/tls.crt"
    keyfile: str = "/var/run/cloudcoil/tls/tls.key"
    host: str = "0.0.0.0"
    port: int = 9443
    service_port: int = 443
    startup_timeout: float = 30
    shutdown_timeout: float = 10

    def __post_init__(self) -> None:
        for port in (self.port, self.service_port):
            if isinstance(port, bool) or not isinstance(port, int) or not 1 <= port <= 65535:
                raise ValueError("Webhook ports must be integers between 1 and 65535")
        for timeout in (self.startup_timeout, self.shutdown_timeout):
            if not math.isfinite(timeout) or timeout <= 0:
                raise ValueError("Webhook timeouts must be finite and positive")

Custom resources

Generate single-version Kubernetes CRDs from the models used by controllers.

Pydantic validators are Python code: use a validation webhook or explicit CEL rules for checks that cannot be represented by the model's JSON Schema.

CRD

A single served/storage version derived from a controller Resource model.

The plural is explicit; group, version and kind come from resource.gvk(). Status is enabled automatically when the model has an optional status field. Passing columns=[] disables the default Age column. No cluster writes occur.

Source code in cloudcoil/crd.py
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
class CRD:
    """A single served/storage version derived from a controller Resource model.

    The plural is explicit; group, version and kind come from ``resource.gvk()``.
    Status is enabled automatically when the model has an optional status field.
    Passing ``columns=[]`` disables the default Age column. No cluster writes occur.
    """

    def __init__(
        self,
        resource: type[Resource],
        *,
        plural: str | None = None,
        scope: Literal["Namespaced", "Cluster"] | None = None,
        singular: str | None = None,
        short_names: Sequence[str] | None = None,
        categories: Sequence[str] | None = None,
        status: bool | None = None,
        columns: Sequence[PrinterColumn] | None = None,
    ) -> None:
        if not isinstance(resource, type) or not issubclass(resource, Resource):
            raise TypeError("resource must be a Resource subclass")
        options = _resource_options(resource)
        plural = plural if plural is not None else options.plural if options else None
        if plural is None:
            raise ValueError("Pass plural=... or declare @custom_resource(plural=...) on the model")
        scope = scope if scope is not None else options.scope if options else "Namespaced"
        singular = singular if singular is not None else options.singular if options else None
        short_names = (
            short_names if short_names is not None else options.short_names if options else ()
        )
        categories = categories if categories is not None else options.categories if options else ()
        status = status if status is not None else options.status if options else None
        columns = columns if columns is not None else options.columns if options else None
        gvk = resource.gvk()
        group, version, kind = gvk.group, gvk.version, gvk.kind
        if gvk.api_version != f"{group}/{version}":
            raise ValueError("api_version must have exactly one group/version separator")
        if (
            not group
            or len(group) > 253
            or any(len(label) > 63 or not _LABEL.fullmatch(label) for label in group.split("."))
        ):
            raise ValueError("CRDs require an api_version with a valid DNS group and version")
        if not _NAME.fullmatch(version) or len(version) > 63:
            raise ValueError("CRD version must be a DNS label starting with a letter")
        if not _KIND.fullmatch(kind):
            raise ValueError("CRD kind must be an alphanumeric identifier starting with a letter")
        singular = singular if singular is not None else kind.lower()
        for label, value in (("plural", plural), ("singular", singular)):
            if not _NAME.fullmatch(value) or len(value) > 63:
                raise ValueError(f"{label} must be a DNS label starting with a letter")
        if len(f"{plural}.{group}") > 253:
            raise ValueError("CRD metadata.name exceeds 253 characters")
        if scope not in ("Namespaced", "Cluster"):
            raise ValueError("scope must be Namespaced or Cluster")
        for label, values in (("short_names", short_names), ("categories", categories)):
            if isinstance(values, str) or any(
                len(value) > 63 or not _LABEL.fullmatch(value) for value in values
            ):
                raise ValueError(f"{label} must contain DNS label names")
        _check_models(resource, set())
        api_field = resource.model_fields["api_version"]
        if (api_field.serialization_alias or api_field.alias) != "apiVersion":
            raise SchemaError("api_version must declare Field(default=..., alias='apiVersion')")
        raw = copy.deepcopy(resource.model_json_schema(by_alias=True, mode="validation"))
        definitions = raw.pop("$defs", {})
        properties = raw.get("properties", {})
        if "kind" not in properties or "metadata" not in properties:
            raise SchemaError("kind and metadata must retain their Kubernetes field names")
        properties["apiVersion"] = {"type": "string"}
        properties["kind"] = {"type": "string"}
        properties["metadata"] = {"type": "object"}
        schema = _convert(raw, definitions, resource.__name__)
        annotated_columns = _printer_columns(schema, enabled=columns is None)
        enabled = "status" in properties if status is None else status
        if enabled:
            if "status" not in properties:
                raise SchemaError("status=True requires a status field on the resource model")
            if "status" in schema.get("required", []):
                raise SchemaError(
                    "status must be optional (for example Status | None = None) because the status subresource strips it on create"
                )
            if schema["properties"]["status"].get("type") != "object":
                raise SchemaError("status must be an object or optional object")
        version_spec: dict[str, Any] = {
            "name": version,
            "served": True,
            "storage": True,
            "schema": {"openAPIV3Schema": schema},
        }
        if enabled:
            version_spec["subresources"] = {"status": {}}
        if columns is None:
            columns = [
                *annotated_columns,
                PrinterColumn(name="Age", json_path=".metadata.creationTimestamp", type="date"),
            ]
        if columns:
            version_spec["additionalPrinterColumns"] = [column._manifest() for column in columns]
        names: dict[str, Any] = {
            "plural": plural,
            "singular": singular,
            "kind": kind,
            "listKind": f"{kind}List",
        }
        if short_names:
            names["shortNames"] = list(short_names)
        if categories:
            names["categories"] = list(categories)
        self.resource = resource
        self.plural = plural
        self.scope = scope
        self._manifest: dict[str, Any] = {
            "apiVersion": "apiextensions.k8s.io/v1",
            "kind": "CustomResourceDefinition",
            "metadata": {"name": f"{plural}.{group}"},
            "spec": {"group": group, "scope": scope, "names": names, "versions": [version_spec]},
        }

    def manifest(self) -> dict[str, Any]:
        """Return an independent manifest suitable for YAML export or API submission."""
        return copy.deepcopy(self._manifest)

    def to_yaml(self) -> str:
        """Serialize the CRD as one YAML document without Python-specific tags."""
        return yaml.safe_dump(self._manifest, sort_keys=False)

manifest()

Return an independent manifest suitable for YAML export or API submission.

Source code in cloudcoil/crd.py
625
626
627
def manifest(self) -> dict[str, Any]:
    """Return an independent manifest suitable for YAML export or API submission."""
    return copy.deepcopy(self._manifest)

to_yaml()

Serialize the CRD as one YAML document without Python-specific tags.

Source code in cloudcoil/crd.py
629
630
631
def to_yaml(self) -> str:
    """Serialize the CRD as one YAML document without Python-specific tags."""
    return yaml.safe_dump(self._manifest, sort_keys=False)

PrinterColumn dataclass

A kubectl column; Annotated fields infer the wire path and scalar type.

For explicit CRD columns supply json_path; type defaults to string there.

Source code in cloudcoil/crd.py
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
@dataclass(frozen=True)
class PrinterColumn:
    """A kubectl column; Annotated fields infer the wire path and scalar type.

    For explicit CRD columns supply json_path; type defaults to string there.
    """

    name: str
    json_path: str | None = None
    type: Literal["string", "integer", "number", "boolean", "date"] | None = None
    description: str | None = None
    format: str | None = None
    priority: int = 0

    def __post_init__(self) -> None:
        if not isinstance(self.name, str) or not self.name:
            raise ValueError("PrinterColumn name must not be empty")
        if self.json_path is not None and (
            not isinstance(self.json_path, str)
            or not self.json_path.startswith(".")
            or any(char in self.json_path for char in "\n\r")
        ):
            raise ValueError("json_path must start with '.' and contain no newline")
        if self.type not in (None, "string", "integer", "number", "boolean", "date"):
            raise ValueError("PrinterColumn type must be string, integer, number, boolean, or date")
        if (
            isinstance(self.priority, bool)
            or not isinstance(self.priority, int)
            or self.priority < 0
        ):
            raise ValueError("PrinterColumn priority must be a nonnegative integer")

    def _manifest(self) -> dict[str, Any]:
        if self.json_path is None:
            raise ValueError("json_path is required outside an Annotated field")
        result = {key: value for key, value in asdict(self).items() if value is not None}
        result.setdefault("type", "string")
        result["jsonPath"] = result.pop("json_path")
        return result

    def __get_pydantic_json_schema__(
        self, schema: CoreSchema, handler: GetJsonSchemaHandler
    ) -> JsonSchemaValue:
        result = handler(schema).copy()
        if "x-cloudcoil-printer-column" in result:
            raise SchemaError("A field can declare only one PrinterColumn")
        result["x-cloudcoil-printer-column"] = {
            key: value for key, value in asdict(self).items() if value is not None
        }
        return result

CEL dataclass

An Annotated constraint enforced by Kubernetes, not by Python validators.

Source code in cloudcoil/crd.py
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
@dataclass(frozen=True)
class CEL:
    """An Annotated constraint enforced by Kubernetes, not by Python validators."""

    rule: str
    message: str | None = None

    def __post_init__(self) -> None:
        if not isinstance(self.rule, str) or not self.rule.strip():
            raise ValueError("CEL rule must not be empty")

    def __get_pydantic_json_schema__(
        self, schema: CoreSchema, handler: GetJsonSchemaHandler
    ) -> JsonSchemaValue:
        result = handler(schema).copy()
        rule = {"rule": self.rule}
        if self.message is not None:
            rule["message"] = self.message
        result["x-kubernetes-validations"] = [*result.get("x-kubernetes-validations", []), rule]
        return result

ListType dataclass

Kubernetes list merge semantics for an Annotated list field.

Map keys refer to serialized field names, just like Kubernetes manifests.

Source code in cloudcoil/crd.py
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
@dataclass(frozen=True)
class ListType:
    """Kubernetes list merge semantics for an Annotated list field.

    Map keys refer to serialized field names, just like Kubernetes manifests.
    """

    type: Literal["atomic", "set", "map"]
    keys: tuple[str, ...] = ()

    def __post_init__(self) -> None:
        if self.type not in ("atomic", "set", "map"):
            raise ValueError("ListType must be atomic, set, or map")
        if isinstance(self.keys, str) or any(
            not isinstance(key, str) or not key for key in self.keys
        ):
            raise ValueError("ListType keys must be nonempty field names")
        object.__setattr__(self, "keys", tuple(self.keys))
        if self.type == "map":
            if not self.keys or len(set(self.keys)) != len(self.keys):
                raise ValueError("Map lists require nonempty, unique keys")
        elif self.keys:
            raise ValueError("Only map lists accept keys")

    def __get_pydantic_json_schema__(
        self, schema: CoreSchema, handler: GetJsonSchemaHandler
    ) -> JsonSchemaValue:
        result = handler(schema).copy()
        if "x-kubernetes-list-type" in result:
            raise SchemaError("A field can declare only one list type")
        result["x-kubernetes-list-type"] = self.type
        if self.keys:
            result["x-kubernetes-list-map-keys"] = list(self.keys)
        return result

SchemaError

Bases: ValueError

A model cannot be represented faithfully as a structural CRD schema.

Source code in cloudcoil/crd.py
25
26
class SchemaError(ValueError):
    """A model cannot be represented faithfully as a structural CRD schema."""

custom_resource(*, plural, api_version=None, kind=None, scope='Namespaced', singular=None, short_names=(), categories=(), status=None, columns=None)

Keep CRD metadata with a Resource; generate with CRD(Model).

The decorator preserves the class and performs no cluster I/O or app registration. Supply api_version to define the wire fields here; kind defaults to the class name. Explicit Literal fields remain supported for precise static typing.

Source code in cloudcoil/crd.py
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
def custom_resource[T: Resource](
    *,
    plural: str,
    api_version: str | None = None,
    kind: str | None = None,
    scope: Literal["Namespaced", "Cluster"] = "Namespaced",
    singular: str | None = None,
    short_names: Sequence[str] = (),
    categories: Sequence[str] = (),
    status: bool | None = None,
    columns: Sequence[PrinterColumn] | None = None,
) -> Callable[[type[T]], type[T]]:
    """Keep CRD metadata with a Resource; generate with CRD(Model).

    The decorator preserves the class and performs no cluster I/O or app registration.
    Supply api_version to define the wire fields here; kind defaults to the class
    name. Explicit Literal fields remain supported for precise static typing.
    """
    if isinstance(short_names, str) or isinstance(categories, str):
        raise ValueError("short_names and categories must be sequences of names")
    options = _ResourceOptions(
        plural,
        scope,
        singular,
        tuple(short_names),
        tuple(categories),
        status,
        tuple(columns) if columns is not None else None,
    )

    def decorate(resource: type[T]) -> type[T]:
        if not isinstance(resource, type) or not issubclass(resource, Resource):
            raise TypeError("custom_resource requires a Resource subclass")
        if _resource_options(resource) is not None:
            raise ValueError("custom_resource metadata is already declared on this class")
        if kind is not None and api_version is None:
            raise ValueError("kind requires api_version")
        if api_version is not None:
            for name, value in (("api_version", api_version), ("kind", kind or resource.__name__)):
                field = copy.copy(resource.model_fields[name])
                if name in resource.__annotations__ and field.default != value:
                    raise ValueError(f"{name} conflicts with the resource field")
                field.annotation = cast(Any, Literal[value])
                field.default = value
                resource.model_fields[name] = field
            resource.model_rebuild(force=True)
        resource.__cloudcoil_crd__ = options  # type: ignore[attr-defined]
        return resource

    return decorate

Admission

Typed, side-effect-free Kubernetes admission webhooks.

AdmissionWebhook

Register typed async mutators/validators and serve them as an ASGI app.

Use an ASGI server for HTTPS and deployment. Callbacks must be side-effect free. An explicit Config enables injected clients for lookups; returned mutations are applied by the API server. Register all routes before serving or generating configurations.

Source code in cloudcoil/admission/_webhook.py
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
class AdmissionWebhook:
    """Register typed async mutators/validators and serve them as an ASGI app.

    Use an ASGI server for HTTPS and deployment. Callbacks must be side-effect
    free. An explicit Config enables injected clients for lookups; returned
    mutations are applied by the API server. Register all routes before serving
    or generating configurations.
    """

    def __init__(
        self, *, config: "Config | None" = None, max_body_bytes: int = 4 * 1024 * 1024
    ) -> None:
        if (
            isinstance(max_body_bytes, bool)
            or not isinstance(max_body_bytes, int)
            or max_body_bytes < 1
        ):
            raise ValueError("max_body_bytes must be a positive integer")
        self._config = config
        self._max_body_bytes = max_body_bytes
        self._routes: dict[str, _Route] = {}

    def register(self, *models: type[Resource]) -> Self:
        """Register policies declared on @custom_resource models, atomically.

        Paths default to /{mutate|validate}/{group}/{version}/{plural}/{method}.
        The optional Config and its HTTP clients remain owned by the caller.
        Client-taking handlers require Config; no discovery occurs during registration.
        """
        return self._register_models(models, require_config=True)

    def _register_models(self, models: Sequence[type[Resource]], *, require_config: bool) -> Self:
        # Application manifest generation can collect routes before loading credentials.
        from cloudcoil.crd import _resource_options

        staged = AdmissionWebhook(config=self._config, max_body_bytes=self._max_body_bytes)
        staged._routes = dict(self._routes)
        for model in models:
            options = _resource_options(model)
            if options is None:
                raise ValueError(
                    f"{model.__name__} needs @custom_resource(plural=...) for registration"
                )
            methods = _methods(model)
            if not methods:
                raise ValueError(f"{model.__name__} has no decorated admission methods")
            gvk = model.gvk()
            for name, handler, policy, needs_client in methods:
                if needs_client and self._config is None and require_config:
                    raise ValueError(f"{model.__name__}.{name} needs AdmissionWebhook(config=...)")

                # Capture each handler independently; route dispatch never closes over loop variables.
                def bind(
                    callback: Callable[..., Awaitable[Any]],
                    resource: type[Resource],
                    inject_client: bool,
                ) -> Callable[[AdmissionRequest[Any]], Awaitable[Any]]:
                    async def invoke(request: AdmissionRequest[Any]) -> Any:
                        if inject_client:
                            client = await request.client(resource)
                            return await callback(request, client)
                        return await callback(request)

                    return invoke

                prefix = "mutate" if policy.mutation else "validate"
                # Stable DNS-label paths also work as Kubernetes Service paths.
                group_path = gvk.group.replace(".", "/")
                method_path = re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-") or "handler"
                path = (
                    policy.path
                    or f"/{prefix}/{group_path}/{gvk.version}/{options.plural}/{method_path}"
                )
                staged._register(
                    model,
                    bind(handler, model, needs_client),
                    policy.mutation,
                    path,
                    options.plural,
                    policy.operations,
                    policy.subresource,
                    options.scope,
                    policy.timeout_seconds,
                    policy.failure_policy,
                )
        self._routes = staged._routes
        return self

    def mutating[T: Resource](
        self,
        model: type[T],
        *,
        target: type[Resource] | None = None,
        resource: str | None = None,
        path: str,
        operations: Sequence[Operation] = ("CREATE", "UPDATE"),
        subresource: str = "",
        scope: Literal["Namespaced", "Cluster", "*"] | None = None,
        namespace_selector: dict[str, Any] | None = None,
        timeout_seconds: int = 5,
        failure_policy: Literal["Fail", "Ignore"] = "Fail",
    ) -> Callable[[Mutator[T]], Mutator[T]]:
        """Register a mutator that returns its edited resource, or None for no patch."""

        def register(handler: Mutator[T]) -> Mutator[T]:
            self._register(
                model,
                handler,
                True,
                path,
                resource,
                operations,
                subresource,
                scope,
                timeout_seconds,
                failure_policy,
                target=target,
                namespace_selector=namespace_selector,
            )
            return handler

        return register

    def validating[T: Resource](
        self,
        model: type[T],
        *,
        target: type[Resource] | None = None,
        resource: str | None = None,
        path: str,
        operations: Sequence[Operation] = ("CREATE", "UPDATE"),
        subresource: str = "",
        scope: Literal["Namespaced", "Cluster", "*"] | None = None,
        namespace_selector: dict[str, Any] | None = None,
        timeout_seconds: int = 5,
        failure_policy: Literal["Fail", "Ignore"] = "Fail",
    ) -> Callable[[Validator[T]], Validator[T]]:
        """Register a validator; return None to allow or raise AdmissionDenied."""

        def register(handler: Validator[T]) -> Validator[T]:
            self._register(
                model,
                handler,
                False,
                path,
                resource,
                operations,
                subresource,
                scope,
                timeout_seconds,
                failure_policy,
                target=target,
                namespace_selector=namespace_selector,
            )
            return handler

        return register

    def _register(
        self,
        model: type[Resource],
        handler: Callable[..., Awaitable[Any]],
        mutation: bool,
        path: str,
        resource: str | None,
        operations: Sequence[Operation],
        subresource: str,
        scope: Literal["Namespaced", "Cluster", "*"] | None,
        timeout_seconds: int,
        failure_policy: Literal["Fail", "Ignore"],
        *,
        target: type[Resource] | None = None,
        namespace_selector: dict[str, Any] | None = None,
    ) -> None:
        from cloudcoil.crd import _resource_options

        model.gvk()  # Fail early for models without a concrete GVK.
        if target is not None and not subresource:
            raise ValueError("A different admission target requires subresource=...")
        target_model = target if target is not None else model
        target_model.gvk()
        options = _resource_options(target_model)
        api = target_model.__dict__.get("__cloudcoil_api__") or {}
        resource = (
            resource if resource is not None else (options.plural if options else api.get("plural"))
        )
        scope = (
            scope if scope is not None else (options.scope if options else api.get("scope", "*"))
        )
        if resource is None:
            raise ValueError("Supply resource=... or use a model with generated API metadata")
        field = model.model_fields["api_version"]
        if (field.serialization_alias or field.alias) != "apiVersion":
            raise ValueError("The resource api_version field needs Field(alias='apiVersion')")
        if not re.fullmatch(r"/[a-z0-9/.-]+", path) or path in self._routes or path == "/healthz":
            raise ValueError(
                "Webhook paths must be unique absolute paths without query or escape characters"
            )
        # Kubernetes validates each Service path segment as a DNS subdomain.
        for segment in path[1:].split("/"):
            _dns(segment)
        if len(resource) > 63 or not re.fullmatch(r"[a-z](?:[-a-z0-9]*[a-z0-9])?", resource):
            raise ValueError("resource must be the exact lowercase Kubernetes resource plural")
        if subresource and not re.fullmatch(r"[a-z][a-z0-9]*", subresource):
            raise ValueError("subresource must be an exact lowercase subresource name")
        if (
            not operations
            or len(set(operations)) != len(operations)
            or any(op not in ("CREATE", "UPDATE", "DELETE") for op in operations)
        ):
            raise ValueError("operations must contain unique CREATE, UPDATE, or DELETE entries")
        if scope not in ("Namespaced", "Cluster", "*") or failure_policy not in ("Fail", "Ignore"):
            raise ValueError("Invalid webhook scope or failure policy")
        if (
            isinstance(timeout_seconds, bool)
            or not isinstance(timeout_seconds, int)
            or not 1 <= timeout_seconds <= 30
        ):
            raise ValueError("timeout_seconds must be an integer between 1 and 30")
        self._routes[path] = _Route(
            model,
            handler,
            mutation,
            path,
            resource,
            tuple(operations),
            subresource,
            scope,
            timeout_seconds,
            failure_policy,
            target,
            deepcopy(namespace_selector),
        )

    def configurations(
        self,
        *,
        name: str,
        service_name: str,
        service_namespace: str,
        ca_bundle: bytes,
        service_port: int = 443,
    ) -> list[dict[str, Any]]:
        """Generate v1 webhook configurations from registered routes.

        ca_bundle is PEM bytes (not base64); Kubernetes needs a certificate valid
        for service_name.service_namespace.svc. This method only returns manifests.
        """
        _dns(name)
        _dns(service_name, label=True)
        _dns(service_namespace, label=True)
        if name.endswith(".static.k8s.io"):
            raise ValueError("The .static.k8s.io suffix is reserved by Kubernetes")
        if not isinstance(ca_bundle, bytes) or b"-----BEGIN CERTIFICATE-----" not in ca_bundle:
            raise ValueError("ca_bundle must contain PEM certificate bytes")
        if (
            isinstance(service_port, bool)
            or not isinstance(service_port, int)
            or not 1 <= service_port <= 65535
        ):
            raise ValueError("service_port must be an integer between 1 and 65535")
        configurations: list[dict[str, Any]] = []
        for mutation, kind in (
            (True, "MutatingWebhookConfiguration"),
            (False, "ValidatingWebhookConfiguration"),
        ):
            webhooks: list[dict[str, Any]] = []
            for index, route in enumerate(self._routes.values()):
                if route.mutation != mutation:
                    continue
                gvk = (route.target or route.model).gvk()
                webhook_name = f"{'mutate' if mutation else 'validate'}-{index}.{name}"
                _dns(webhook_name)
                if webhook_name.count(".") < 2:
                    raise ValueError("name must include a domain, such as widgets.example.com")
                webhook: dict[str, Any] = {
                    "name": webhook_name,
                    "admissionReviewVersions": ["v1"],
                    "sideEffects": "None",
                    "failurePolicy": route.failure_policy,
                    "matchPolicy": "Exact",
                    "timeoutSeconds": route.timeout_seconds,
                    "clientConfig": {
                        "service": {
                            "name": service_name,
                            "namespace": service_namespace,
                            "path": route.path,
                            "port": service_port,
                        },
                        "caBundle": base64.b64encode(ca_bundle).decode("ascii"),
                    },
                    "rules": [
                        {
                            "apiGroups": [gvk.group],
                            "apiVersions": [gvk.version],
                            "resources": [
                                route.resource
                                + (f"/{route.subresource}" if route.subresource else "")
                            ],
                            "operations": list(route.operations),
                            "scope": route.scope,
                        }
                    ],
                }
                if route.namespace_selector is not None:
                    webhook["namespaceSelector"] = deepcopy(route.namespace_selector)
                if mutation:
                    webhook["reinvocationPolicy"] = "Never"
                webhooks.append(webhook)
            if webhooks:
                configurations.append(
                    {
                        "apiVersion": "admissionregistration.k8s.io/v1",
                        "kind": kind,
                        "metadata": {"name": name},
                        "webhooks": webhooks,
                    }
                )
        return configurations

    async def __call__(self, scope: dict[str, Any], receive: Receive, send: Send) -> None:
        if scope["type"] == "lifespan":
            while True:
                message = await receive()
                if message["type"] == "lifespan.startup":
                    await send({"type": "lifespan.startup.complete"})
                elif message["type"] == "lifespan.shutdown":
                    await send({"type": "lifespan.shutdown.complete"})
                    return
        if scope["type"] != "http":
            raise ValueError("AdmissionWebhook only supports HTTP and lifespan ASGI scopes")
        if scope.get("path") == "/healthz" and scope.get("method") == "GET":
            await self._respond(send, 200, {"status": "ok"})
            return
        route = self._routes.get(scope.get("path", ""))
        if route is None:
            await self._respond(send, 404, {"message": "Unknown webhook path"})
            return
        if scope.get("method") != "POST":
            await self._respond(send, 405, {"message": "Webhook requests require POST"})
            return
        headers = {key.lower(): value for key, value in scope.get("headers", [])}
        if (
            headers.get(b"content-type", b"").split(b";", 1)[0].strip().lower()
            != b"application/json"
        ):
            await self._respond(send, 415, {"message": "Expected application/json"})
            return
        response_started = False

        async def respond(status: int, value: dict[str, Any]) -> None:
            nonlocal response_started
            response_started = True
            await self._respond(send, status, value)

        try:
            async with asyncio.timeout(route.timeout_seconds):
                body = bytearray()
                while True:
                    message = await receive()
                    if message["type"] == "http.disconnect":
                        return
                    if message["type"] != "http.request":
                        raise ValueError("Unexpected ASGI request event")
                    chunk = message.get("body", b"")
                    if len(body) + len(chunk) > self._max_body_bytes:
                        await respond(413, {"message": "AdmissionReview exceeds max_body_bytes"})
                        return
                    body.extend(chunk)
                    if not message.get("more_body", False):
                        break
                try:
                    # JSON parsers may accept NaN/Infinity; Kubernetes JSON does not.
                    payload = json.loads(body)
                    json.dumps(payload, allow_nan=False)
                    review = _Review.model_validate(payload)
                    self._check_request(route, review.request)
                except ValidationError, ValueError:
                    await respond(
                        400,
                        {"message": "Invalid or mismatched admission.k8s.io/v1 AdmissionReview"},
                    )
                    return
                try:
                    request = self._request(route, review.request)
                    request._config = self._config
                except ValidationError as error:
                    causes = [
                        {
                            "reason": "FieldValueInvalid",
                            "field": ".".join(str(part) for part in issue["loc"])[:256],
                            "message": issue["msg"][:256],
                        }
                        for issue in error.errors(
                            include_input=False, include_url=False, include_context=False
                        )[:20]
                    ]
                    await respond(
                        200,
                        {
                            "apiVersion": "admission.k8s.io/v1",
                            "kind": "AdmissionReview",
                            "response": {
                                "uid": review.request.uid,
                                "allowed": False,
                                "status": {
                                    "code": 422,
                                    "reason": "Invalid",
                                    "message": "Invalid resource: "
                                    + "; ".join(
                                        f"{cause['field']}: {cause['message']}"
                                        for cause in causes[:3]
                                    ),
                                    "details": {"causes": causes},
                                },
                            },
                        },
                    )
                    return
                # Keep the baseline and raw JSON separate from callback-owned data.
                before = request.resource.model_copy(deep=True) if request.resource else None
                response: dict[str, Any] = {"uid": request.uid, "allowed": True}
                try:
                    connected, result = await self._invoke(route, request, receive)
                    if not connected:
                        return
                    if result is not None:
                        if not route.mutation or not isinstance(result, route.model):
                            raise TypeError(
                                "Validators return None; mutators return their registered resource type or None"
                            )
                        if before is None or review.request.object is None:
                            raise ValueError(
                                "Cannot mutate an admission request without a current object"
                            )
                        patch = mutation_patch(review.request.object, before, result)
                        if patch:
                            response.update(
                                patchType="JSONPatch",
                                patch=base64.b64encode(
                                    json.dumps(
                                        patch, allow_nan=False, separators=(",", ":")
                                    ).encode()
                                ).decode("ascii"),
                            )
                except AdmissionDenied as denied:
                    response = {
                        "uid": request.uid,
                        "allowed": False,
                        "status": {
                            "code": denied.code,
                            "reason": denied.reason,
                            "message": str(denied),
                        },
                    }
                await respond(
                    200,
                    {
                        "apiVersion": "admission.k8s.io/v1",
                        "kind": "AdmissionReview",
                        "response": response,
                    },
                )
        except TimeoutError:
            if not response_started:
                await respond(504, {"message": "Admission webhook timed out"})
        except Exception:
            if response_started:
                raise
            logger.exception("Admission handler failed: path=%s", route.path)
            await respond(500, {"message": "Admission webhook failed"})

    @staticmethod
    async def _invoke(
        route: _Route, request: AdmissionRequest[Any], receive: Receive
    ) -> tuple[bool, Any]:
        async def invoke() -> Any:
            return await route.handler(request)

        async def disconnected() -> None:
            while True:
                message = await receive()
                if message["type"] == "http.disconnect":
                    return
                if message["type"] != "http.request":
                    raise ValueError("Unexpected ASGI event after request body")

        handler = asyncio.create_task(invoke())
        disconnect = asyncio.create_task(disconnected())
        try:
            done, _ = await asyncio.wait((handler, disconnect), return_when=asyncio.FIRST_COMPLETED)
            if disconnect in done:
                await disconnect
                return False, None
            return True, await handler
        finally:
            for task in (handler, disconnect):
                task.cancel()
            await asyncio.gather(handler, disconnect, return_exceptions=True)

    @staticmethod
    def _check_request(route: _Route, raw: _Request) -> None:
        gvk = route.model.gvk()
        target = (route.target or route.model).gvk()
        if (raw.kind.group, raw.kind.version, raw.kind.kind) != (
            gvk.group,
            gvk.version,
            gvk.kind,
        ) or (raw.resource.group, raw.resource.version, raw.resource.resource) != (
            target.group,
            target.version,
            route.resource,
        ):
            raise ValueError("Admission kind or resource does not match this route")
        if raw.operation not in route.operations or raw.subresource != route.subresource:
            raise ValueError("Admission operation or subresource does not match this route")
        if (route.scope == "Namespaced" and not raw.namespace) or (
            route.scope == "Cluster" and raw.namespace
        ):
            raise ValueError("Admission namespace does not match this route's scope")
        if raw.operation == "DELETE":
            if raw.object is not None or raw.old_object is None:
                raise ValueError("DELETE needs oldObject and no object")
        elif (
            raw.object is None
            or (raw.operation == "CREATE" and raw.old_object is not None)
            or (raw.operation == "UPDATE" and raw.old_object is None)
        ):
            raise ValueError("Invalid object/oldObject for admission operation")
        for value in (raw.object, raw.old_object):
            if value is not None and (value.get("apiVersion"), value.get("kind")) != (
                gvk.api_version,
                gvk.kind,
            ):
                raise ValueError("Object GVK does not match this route")

    @staticmethod
    def _request(route: _Route, raw: _Request) -> AdmissionRequest[Any]:
        return AdmissionRequest[Any](
            uid=raw.uid,
            operation=raw.operation,
            resource=route.model.model_validate(deepcopy(raw.object))
            if raw.object is not None
            else None,
            old_resource=route.model.model_validate(deepcopy(raw.old_object))
            if raw.old_object is not None
            else None,
            name=raw.name,
            namespace=raw.namespace,
            subresource=raw.subresource,
            dry_run=raw.dry_run,
            user_info=raw.user_info.model_copy(deep=True),
            options=deepcopy(raw.options),
            raw_object=deepcopy(raw.object),
            raw_old_object=deepcopy(raw.old_object),
        )

    @staticmethod
    async def _respond(send: Send, status: int, value: dict[str, Any]) -> None:
        body = json.dumps(value, allow_nan=False, separators=(",", ":")).encode("utf-8")
        await send(
            {
                "type": "http.response.start",
                "status": status,
                "headers": [
                    (b"content-type", b"application/json"),
                    (b"content-length", str(len(body)).encode("ascii")),
                ],
            }
        )
        await send({"type": "http.response.body", "body": body})

configurations(*, name, service_name, service_namespace, ca_bundle, service_port=443)

Generate v1 webhook configurations from registered routes.

ca_bundle is PEM bytes (not base64); Kubernetes needs a certificate valid for service_name.service_namespace.svc. This method only returns manifests.

Source code in cloudcoil/admission/_webhook.py
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
def configurations(
    self,
    *,
    name: str,
    service_name: str,
    service_namespace: str,
    ca_bundle: bytes,
    service_port: int = 443,
) -> list[dict[str, Any]]:
    """Generate v1 webhook configurations from registered routes.

    ca_bundle is PEM bytes (not base64); Kubernetes needs a certificate valid
    for service_name.service_namespace.svc. This method only returns manifests.
    """
    _dns(name)
    _dns(service_name, label=True)
    _dns(service_namespace, label=True)
    if name.endswith(".static.k8s.io"):
        raise ValueError("The .static.k8s.io suffix is reserved by Kubernetes")
    if not isinstance(ca_bundle, bytes) or b"-----BEGIN CERTIFICATE-----" not in ca_bundle:
        raise ValueError("ca_bundle must contain PEM certificate bytes")
    if (
        isinstance(service_port, bool)
        or not isinstance(service_port, int)
        or not 1 <= service_port <= 65535
    ):
        raise ValueError("service_port must be an integer between 1 and 65535")
    configurations: list[dict[str, Any]] = []
    for mutation, kind in (
        (True, "MutatingWebhookConfiguration"),
        (False, "ValidatingWebhookConfiguration"),
    ):
        webhooks: list[dict[str, Any]] = []
        for index, route in enumerate(self._routes.values()):
            if route.mutation != mutation:
                continue
            gvk = (route.target or route.model).gvk()
            webhook_name = f"{'mutate' if mutation else 'validate'}-{index}.{name}"
            _dns(webhook_name)
            if webhook_name.count(".") < 2:
                raise ValueError("name must include a domain, such as widgets.example.com")
            webhook: dict[str, Any] = {
                "name": webhook_name,
                "admissionReviewVersions": ["v1"],
                "sideEffects": "None",
                "failurePolicy": route.failure_policy,
                "matchPolicy": "Exact",
                "timeoutSeconds": route.timeout_seconds,
                "clientConfig": {
                    "service": {
                        "name": service_name,
                        "namespace": service_namespace,
                        "path": route.path,
                        "port": service_port,
                    },
                    "caBundle": base64.b64encode(ca_bundle).decode("ascii"),
                },
                "rules": [
                    {
                        "apiGroups": [gvk.group],
                        "apiVersions": [gvk.version],
                        "resources": [
                            route.resource
                            + (f"/{route.subresource}" if route.subresource else "")
                        ],
                        "operations": list(route.operations),
                        "scope": route.scope,
                    }
                ],
            }
            if route.namespace_selector is not None:
                webhook["namespaceSelector"] = deepcopy(route.namespace_selector)
            if mutation:
                webhook["reinvocationPolicy"] = "Never"
            webhooks.append(webhook)
        if webhooks:
            configurations.append(
                {
                    "apiVersion": "admissionregistration.k8s.io/v1",
                    "kind": kind,
                    "metadata": {"name": name},
                    "webhooks": webhooks,
                }
            )
    return configurations

mutating(model, *, target=None, resource=None, path, operations=('CREATE', 'UPDATE'), subresource='', scope=None, namespace_selector=None, timeout_seconds=5, failure_policy='Fail')

Register a mutator that returns its edited resource, or None for no patch.

Source code in cloudcoil/admission/_webhook.py
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
def mutating[T: Resource](
    self,
    model: type[T],
    *,
    target: type[Resource] | None = None,
    resource: str | None = None,
    path: str,
    operations: Sequence[Operation] = ("CREATE", "UPDATE"),
    subresource: str = "",
    scope: Literal["Namespaced", "Cluster", "*"] | None = None,
    namespace_selector: dict[str, Any] | None = None,
    timeout_seconds: int = 5,
    failure_policy: Literal["Fail", "Ignore"] = "Fail",
) -> Callable[[Mutator[T]], Mutator[T]]:
    """Register a mutator that returns its edited resource, or None for no patch."""

    def register(handler: Mutator[T]) -> Mutator[T]:
        self._register(
            model,
            handler,
            True,
            path,
            resource,
            operations,
            subresource,
            scope,
            timeout_seconds,
            failure_policy,
            target=target,
            namespace_selector=namespace_selector,
        )
        return handler

    return register

register(*models)

Register policies declared on @custom_resource models, atomically.

Paths default to /{mutate|validate}/{group}/{version}/{plural}/{method}. The optional Config and its HTTP clients remain owned by the caller. Client-taking handlers require Config; no discovery occurs during registration.

Source code in cloudcoil/admission/_webhook.py
120
121
122
123
124
125
126
127
def register(self, *models: type[Resource]) -> Self:
    """Register policies declared on @custom_resource models, atomically.

    Paths default to /{mutate|validate}/{group}/{version}/{plural}/{method}.
    The optional Config and its HTTP clients remain owned by the caller.
    Client-taking handlers require Config; no discovery occurs during registration.
    """
    return self._register_models(models, require_config=True)

validating(model, *, target=None, resource=None, path, operations=('CREATE', 'UPDATE'), subresource='', scope=None, namespace_selector=None, timeout_seconds=5, failure_policy='Fail')

Register a validator; return None to allow or raise AdmissionDenied.

Source code in cloudcoil/admission/_webhook.py
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
def validating[T: Resource](
    self,
    model: type[T],
    *,
    target: type[Resource] | None = None,
    resource: str | None = None,
    path: str,
    operations: Sequence[Operation] = ("CREATE", "UPDATE"),
    subresource: str = "",
    scope: Literal["Namespaced", "Cluster", "*"] | None = None,
    namespace_selector: dict[str, Any] | None = None,
    timeout_seconds: int = 5,
    failure_policy: Literal["Fail", "Ignore"] = "Fail",
) -> Callable[[Validator[T]], Validator[T]]:
    """Register a validator; return None to allow or raise AdmissionDenied."""

    def register(handler: Validator[T]) -> Validator[T]:
        self._register(
            model,
            handler,
            False,
            path,
            resource,
            operations,
            subresource,
            scope,
            timeout_seconds,
            failure_policy,
            target=target,
            namespace_selector=namespace_selector,
        )
        return handler

    return register

AdmissionRequest

Bases: BaseModel

A typed snapshot of one admission request; DELETE has resource=None.

Callbacks must have no external side effects, including during dry runs. Return an edited resource from a mutator; editing this snapshot and returning None has no effect. raw_object and raw_old_object are copies for inspection.

Source code in cloudcoil/admission/_types.py
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
class AdmissionRequest[T: Resource](BaseModel):
    """A typed snapshot of one admission request; DELETE has resource=None.

    Callbacks must have no external side effects, including during dry runs.
    Return an edited resource from a mutator; editing this snapshot and returning
    None has no effect. raw_object and raw_old_object are copies for inspection.
    """

    model_config = ConfigDict(frozen=True)

    _config: "Config | None" = PrivateAttr(default=None)

    @property
    def config(self) -> "Config | None":
        """Caller-owned config injected by AdmissionWebhook, for other resource clients."""
        return self._config

    async def client[U: Resource](self, resource: type[U]) -> "AsyncAPIClient[U]":
        """Read any resource kind using the webhook's connection and request namespace.

        Admission handlers must only read: API writes would have side effects
        even when the admission request is a dry run or is subsequently rejected.
        """
        if self._config is None:
            raise RuntimeError("Admission clients require AdmissionWebhook(config=...)")
        return await resource.async_client(
            self._config, namespace=self.namespace or None, cached=False
        )

    def cached[U: Resource](self, resource: type[U]) -> CachedResources[U]:
        """Read an explicitly preconfigured Config cache, never a leader's cache.

        Admission serves on every replica. Declare Cache(resources=[...]) on
        Config, with wait_for_sync=True and mode='strict'. Staleness remains;
        use client() for checks requiring a live read.
        """
        if (
            self._config is None
            or not self._config.cache.enabled
            or resource not in (self._config.cache.resources or [])
        ):
            raise ValueError("Admission cached reads require a preconfigured Config cache resource")
        from cloudcoil.caching import AsyncInformer

        informer = self._config.cache.get_informer(resource, sync=False)
        if not isinstance(informer, AsyncInformer):
            raise RuntimeError("Admission requires an async informer")
        return CachedResources(informer, self.namespace or None)

    uid: str
    operation: Operation
    resource: T | None
    old_resource: T | None
    name: str = ""
    namespace: str = ""
    subresource: str = ""
    dry_run: bool = False
    user_info: UserInfo = Field(default_factory=UserInfo)
    options: dict[str, Any] | None = None
    raw_object: dict[str, Any] | None = None
    raw_old_object: dict[str, Any] | None = None

config property

Caller-owned config injected by AdmissionWebhook, for other resource clients.

cached(resource)

Read an explicitly preconfigured Config cache, never a leader's cache.

Admission serves on every replica. Declare Cache(resources=[...]) on Config, with wait_for_sync=True and mode='strict'. Staleness remains; use client() for checks requiring a live read.

Source code in cloudcoil/admission/_types.py
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
def cached[U: Resource](self, resource: type[U]) -> CachedResources[U]:
    """Read an explicitly preconfigured Config cache, never a leader's cache.

    Admission serves on every replica. Declare Cache(resources=[...]) on
    Config, with wait_for_sync=True and mode='strict'. Staleness remains;
    use client() for checks requiring a live read.
    """
    if (
        self._config is None
        or not self._config.cache.enabled
        or resource not in (self._config.cache.resources or [])
    ):
        raise ValueError("Admission cached reads require a preconfigured Config cache resource")
    from cloudcoil.caching import AsyncInformer

    informer = self._config.cache.get_informer(resource, sync=False)
    if not isinstance(informer, AsyncInformer):
        raise RuntimeError("Admission requires an async informer")
    return CachedResources(informer, self.namespace or None)

client(resource) async

Read any resource kind using the webhook's connection and request namespace.

Admission handlers must only read: API writes would have side effects even when the admission request is a dry run or is subsequently rejected.

Source code in cloudcoil/admission/_types.py
43
44
45
46
47
48
49
50
51
52
53
async def client[U: Resource](self, resource: type[U]) -> "AsyncAPIClient[U]":
    """Read any resource kind using the webhook's connection and request namespace.

    Admission handlers must only read: API writes would have side effects
    even when the admission request is a dry run or is subsequently rejected.
    """
    if self._config is None:
        raise RuntimeError("Admission clients require AdmissionWebhook(config=...)")
    return await resource.async_client(
        self._config, namespace=self.namespace or None, cached=False
    )

AdmissionDenied

Bases: Exception

Reject admission with a message visible to the requesting Kubernetes user.

Source code in cloudcoil/admission/_types.py
89
90
91
92
93
94
95
96
97
98
99
class AdmissionDenied(Exception):
    """Reject admission with a message visible to the requesting Kubernetes user."""

    def __init__(self, message: str, *, code: int = 403, reason: str = "Forbidden") -> None:
        if not message:
            raise ValueError("An admission denial needs a message")
        if isinstance(code, bool) or not isinstance(code, int) or not 400 <= code <= 599:
            raise ValueError("A denial code must be an integer between 400 and 599")
        super().__init__(message)
        self.code = code
        self.reason = reason

UserInfo

Bases: BaseModel

Identity supplied by the Kubernetes API server, not independently authenticated here.

Source code in cloudcoil/admission/_types.py
17
18
19
20
21
22
23
class UserInfo(BaseModel):
    """Identity supplied by the Kubernetes API server, not independently authenticated here."""

    username: str = ""
    uid: str = ""
    groups: list[str] = Field(default_factory=list)
    extra: dict[str, list[str]] = Field(default_factory=dict)

mutating(*, path=None, operations=('CREATE', 'UPDATE'), subresource='', timeout_seconds=5, failure_policy='Fail')

Mark an async Resource class/static method that returns an edited resource.

The bound method takes AdmissionRequest and optionally an injected AsyncAPIClient. Register the model explicitly with AdmissionWebhook.register before serving.

Source code in cloudcoil/admission/_decorators.py
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
def mutating[**P, R](
    *,
    path: str | None = None,
    operations: Sequence[Operation] = ("CREATE", "UPDATE"),
    subresource: str = "",
    timeout_seconds: int = 5,
    failure_policy: Literal["Fail", "Ignore"] = "Fail",
) -> Callable[[Callable[P, Awaitable[R]]], Callable[P, Awaitable[R]]]:
    """Mark an async Resource class/static method that returns an edited resource.

    The bound method takes AdmissionRequest and optionally an injected AsyncAPIClient.
    Register the model explicitly with AdmissionWebhook.register before serving.
    """
    return _decorate(
        _AdmissionMethod(
            True, path, tuple(operations), subresource, timeout_seconds, failure_policy
        )
    )

validating(*, path=None, operations=('CREATE', 'UPDATE'), subresource='', timeout_seconds=5, failure_policy='Fail')

Mark an async Resource class/static method; raise AdmissionDenied to reject.

Source code in cloudcoil/admission/_decorators.py
57
58
59
60
61
62
63
64
65
66
67
68
69
70
def validating[**P, R](
    *,
    path: str | None = None,
    operations: Sequence[Operation] = ("CREATE", "UPDATE"),
    subresource: str = "",
    timeout_seconds: int = 5,
    failure_policy: Literal["Fail", "Ignore"] = "Fail",
) -> Callable[[Callable[P, Awaitable[R]]], Callable[P, Awaitable[R]]]:
    """Mark an async Resource class/static method; raise AdmissionDenied to reject."""
    return _decorate(
        _AdmissionMethod(
            False, path, tuple(operations), subresource, timeout_seconds, failure_policy
        )
    )

Caching and runtime

CloudCoil Caching - Efficient client-side caching for Kubernetes resources.

This module provides a caching system powered by informers that watch Kubernetes resources and maintain a local cache, similar to client-go's informer pattern.

Key components: - Cache: Configuration and control for the informer system - AsyncInformer/SyncInformer: Watch and cache individual resource types - ConcurrentStore: Thread-safe storage with custom indexing - CachedClient/AsyncCachedClient: Client wrappers that use cache for reads

Example usage:

from cloudcoil.caching import Cache
from cloudcoil.client import Config
import cloudcoil.models.kubernetes as k8s

# Basic usage - just enable caching
config = Config(cache=True)

with config:
    deployment = k8s.apps.v1.Deployment.get("my-app")  # From cache

# Advanced configuration
cache_config = Cache(
    resync_period=600,  # 10 minutes
    mode="strict",      # Cache-only mode
    resources=[k8s.apps.v1.Deployment, k8s.core.v1.Pod]
)
config = Config(cache=cache_config)

# Event handling
async with config:
    informer = config.cache.get_informer(k8s.apps.v1.Deployment)

    @informer.on_add
    def handle_new_deployment(deployment):
        print(f"New deployment: {deployment.metadata.name}")

Cache

Bases: Cache

Cache configuration and runtime control.

This extends the CacheConfig Pydantic model with runtime methods for controlling the informer system lifecycle.

Configuration fields are validated by Pydantic, while runtime methods provide lifecycle control following CloudCoil's async-first patterns.

Source code in cloudcoil/caching/_cache.py
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
class Cache(CacheConfig):
    """Cache configuration and runtime control.

    This extends the CacheConfig Pydantic model with runtime methods for
    controlling the informer system lifecycle.

    Configuration fields are validated by Pydantic, while runtime methods
    provide lifecycle control following CloudCoil's async-first patterns.
    """

    def __init__(self, **data: Any) -> None:
        """Initialize cache with configuration."""
        super().__init__(**data)
        self._client_factory: Optional[Any] = None  # Will be set to client_for method
        self._async_informers: Dict[str, "AsyncInformer[Any]"] = {}
        self._sync_informers: Dict[str, "SyncInformer[Any]"] = {}
        self._started: bool = False
        self._start_tasks: List[asyncio.Task] = []  # Track informer start tasks
        self._pause_count: int = 0  # For nested pause context managers
        self._original_enabled: Optional[bool] = None  # Store original state

    def set_client_factory(self, factory: Any) -> None:
        """Set the client factory function (called by Config)."""
        self._client_factory = factory

    @overload
    def _create_client(self, resource_type: Type[T], sync: Literal[True]) -> "APIClient[T]": ...

    @overload
    def _create_client(
        self, resource_type: Type[T], sync: Literal[False]
    ) -> "AsyncAPIClient[T]": ...

    def _create_client(
        self, resource_type: Type[T], sync: bool
    ) -> Union["APIClient[T]", "AsyncAPIClient[T]"]:
        """Create a properly typed client using the factory."""
        if not self._client_factory:
            raise RuntimeError("Client factory not set")
        # Use cached=False to get base client without caching
        return self._client_factory(resource_type, sync=sync, cached=False)  # type: ignore[return-value]

    # Async methods (CloudCoil async_ prefix pattern)
    async def async_start(self) -> None:
        """Start all informers asynchronously."""
        if not self.enabled or self._started:
            return

        self._started = True

        # Pre-create informers for configured resources
        if self.resources:
            logger.debug("Pre-creating informers for %d configured resources", len(self.resources))
            for resource_type in self.resources:
                # This will create the informer if it doesn't exist
                self.get_informer(resource_type, sync=False)

        # Start all async informers and track tasks
        self._start_tasks = []
        if self._async_informers:
            for key, informer in self._async_informers.items():
                if isinstance(informer, AsyncInformer):
                    logger.debug("Starting informer for %s", key)
                    task = asyncio.create_task(informer._start())
                    self._start_tasks.append(task)

            # Wait for all informers to start
            if self._start_tasks:
                await asyncio.gather(*self._start_tasks, return_exceptions=True)

        logger.debug("Cache started with %d async informers", len(self._async_informers))

    async def async_stop(self) -> None:
        """Stop all informers asynchronously."""
        if not self._started:
            return

        self._started = False

        # Stop all async informers
        if self._async_informers:
            tasks = []
            for informer in self._async_informers.values():
                if isinstance(informer, AsyncInformer):
                    tasks.append(informer._stop())
            if tasks:
                await asyncio.gather(*tasks, return_exceptions=True)

        logger.debug("Cache stopped")

    async def async_wait(self, timeout: Optional[float] = None) -> bool:
        """Wait for cache to be ready asynchronously.

        Args:
            timeout: Maximum time to wait (uses sync_timeout if not provided)

        Returns:
            True if ready, False if timeout
        """
        if not self.enabled:
            return True

        # If resources are configured but no informers created yet, that's not ready
        if self.resources and not self._async_informers and not self._sync_informers:
            logger.warning("Cache has configured resources but no informers created yet")
            return False

        # If no informers at all and no resources configured, that's OK
        if not self._async_informers and not self._sync_informers and not self.resources:
            return True

        timeout = self.sync_timeout if timeout is None else timeout

        # First wait for any pending start tasks
        if self._start_tasks:
            try:
                await asyncio.wait_for(
                    asyncio.gather(*self._start_tasks, return_exceptions=True),
                    timeout=timeout / 2,  # Use half timeout for starting
                )
            except asyncio.TimeoutError:
                logger.warning("Timeout waiting for informers to start")
                return False

        # Then wait for all async informers to sync
        sync_tasks = []
        for informer in self._async_informers.values():
            if isinstance(informer, AsyncInformer):
                sync_tasks.append(informer._wait_for_sync(timeout))

        if sync_tasks:
            results = await asyncio.gather(*sync_tasks, return_exceptions=True)
            return all(r is True for r in results)

        return True

    # Synchronous methods (verb-based naming)
    def start(self) -> None:
        """Start all informers synchronously (blocks until started)."""
        if not self.enabled or self._started:
            return

        self._started = True

        # Pre-create informers for configured resources
        if self.resources:
            logger.debug(
                "Pre-creating sync informers for %d configured resources", len(self.resources)
            )
            for resource_type in self.resources:
                # This will create the informer if it doesn't exist
                self.get_informer(resource_type, sync=True)

        # Start all sync informers
        for informer in self._sync_informers.values():
            if isinstance(informer, SyncInformer):
                informer._start()

        logger.debug("Cache started with %d sync informers", len(self._sync_informers))

    def stop(self) -> None:
        """Stop all informers synchronously (blocks until stopped)."""
        if not self._started:
            return

        self._started = False

        # Stop all sync informers
        for informer in self._sync_informers.values():
            if isinstance(informer, SyncInformer):
                informer._stop()

        logger.debug("Cache stopped")

    def wait(self, timeout: Optional[float] = None) -> bool:
        """Wait for cache to be ready synchronously.

        Args:
            timeout: Maximum time to wait (uses sync_timeout if not provided)

        Returns:
            True if ready, False if timeout
        """
        if not self.enabled:
            return True

        # If resources are configured but no informers created yet, that's not ready
        if self.resources and not self._sync_informers and not self._async_informers:
            logger.warning("Cache has configured resources but no informers created yet")
            return False

        # If no informers at all and no resources configured, that's OK
        if not self._sync_informers and not self._async_informers and not self.resources:
            return True

        timeout = self.sync_timeout if timeout is None else timeout

        # Wait for all sync informers to sync
        start_time = time.time()
        for informer in self._sync_informers.values():
            if isinstance(informer, SyncInformer):
                remaining = max(0, timeout - (time.time() - start_time))
                if not informer._wait_for_sync(remaining):
                    return False

        return True

    # Non-blocking methods (no I/O)
    def ready(self) -> bool:
        """Check if cache is synced and ready (non-blocking check)."""
        if not self.enabled:
            return True
        if not self._started:
            return False

        # Check if all informers have synced
        for informer in self._async_informers.values():
            if isinstance(informer, AsyncInformer) and not informer._has_synced():
                return False

        for sync_informer in self._sync_informers.values():
            if isinstance(sync_informer, SyncInformer) and not sync_informer._has_synced():
                return False

        return True

    def get_informer(
        self,
        resource_type: Type[T],
        sync: bool = False,
        namespace: Optional[str] = None,
    ) -> Optional[Union["AsyncInformer[T]", "SyncInformer[T]"]]:
        """Get or create informer for a specific resource type.

        Args:
            resource_type: The resource type
            sync: Whether to return sync wrapper
            namespace: Optional namespace scope

        Returns:
            The informer instance, or None if not available
        """
        if not self.enabled or not self._client_factory:
            return None

        # Check if we should cache this resource
        if not self.should_cache(resource_type):
            return None

        # Create cache key
        key = f"{resource_type.__module__}.{resource_type.__name__}"
        if namespace:
            key += f":{namespace}"

        if sync:
            # Get or create sync informer
            if key not in self._sync_informers:
                # Create sync client using factory
                sync_client: APIClient[T] = self._create_client(resource_type, True)
                options = self._create_options(resource_type, namespace)

                sync_informer: SyncInformer[T] = SyncInformer(sync_client, options)
                self._sync_informers[key] = sync_informer  # type: ignore[assignment]

                # Auto-start if cache is running
                if self._started:
                    sync_informer._start()  # type: ignore[func-returns-value]
                    # Note: Sync informers start synchronously, no task tracking needed

            return self._sync_informers[key]  # type: ignore[return-value]
        else:
            # Get or create async informer
            if key not in self._async_informers:
                # Create async client using factory
                async_client: AsyncAPIClient[T] = self._create_client(resource_type, False)
                options = self._create_options(resource_type, namespace)

                async_informer: AsyncInformer[T] = AsyncInformer(async_client, options)
                self._async_informers[key] = async_informer  # type: ignore[assignment]

                # Auto-start if cache is running
                if self._started:
                    task = asyncio.create_task(async_informer._start())
                    self._start_tasks.append(task)

                    # Also create a task to wait for initial sync
                    async def wait_for_sync():
                        try:
                            await task  # Wait for start to complete
                            await async_informer._wait_for_sync(self.sync_timeout)
                        except Exception as e:
                            logger.error("Error starting informer for %s: %s", key, e)

                    asyncio.create_task(wait_for_sync())

            return self._async_informers[key]  # type: ignore[return-value]

    def _create_options(
        self,
        resource_type: Type["Resource"],
        namespace: Optional[str],
    ) -> InformerOptions:
        """Create informer options for resource type."""
        if self.namespaces:
            if namespace is not None and namespace not in self.namespaces:
                raise ValueError("Requested namespace is outside this cache's configured scope")
            namespace = namespace or self.namespaces[0]
        overrides = (self.per_resource or {}).get(resource_type)
        return InformerOptions(
            resync_period=overrides.resync_period
            if overrides and overrides.resync_period is not None
            else self.resync_period,
            namespace=namespace,
            all_namespaces=namespace is None,
            max_items=overrides.max_items if overrides else self.max_items_per_resource,
            label_selector=overrides.label_selector
            if overrides and overrides.label_selector is not None
            else self.label_selector,
            field_selector=overrides.field_selector
            if overrides and overrides.field_selector is not None
            else self.field_selector,
        )

    # Context managers
    @contextmanager
    def pause(self) -> Generator[None, None, None]:
        """Temporarily disable cache within this context.

        Supports nested usage with reference counting.
        """
        self._pause_count += 1
        if self._pause_count == 1:
            # First pause - save state and disable
            self._original_enabled = self.enabled
            self.enabled = False
        try:
            yield
        finally:
            self._pause_count -= 1
            if self._pause_count == 0:
                # Last pause exiting - restore state
                if self._original_enabled is not None:
                    self.enabled = self._original_enabled
                self._original_enabled = None

    @contextmanager
    def strict_mode(self) -> Generator[None, None, None]:
        """Temporarily use strict mode within this context."""
        original_mode = self.mode
        self.mode = "strict"
        try:
            yield
        finally:
            self.mode = original_mode

    @contextmanager
    def fallback_mode(self) -> Generator[None, None, None]:
        """Temporarily use fallback mode within this context."""
        original_mode = self.mode
        self.mode = "fallback"
        try:
            yield
        finally:
            self.mode = original_mode

    def status(self) -> CacheStatus:
        """Get cache status information."""
        if not self.enabled:
            return CacheStatus(enabled=False)

        if not self._client_factory:
            return CacheStatus(enabled=True, ready=False)

        # Count total resources across all informers
        resource_count = 0
        for informer in self._async_informers.values():
            resource_count += informer._store.size()
        for sync_informer in self._sync_informers.values():
            resource_count += sync_informer._store.size()

        return CacheStatus(
            enabled=True,
            started=self._started,
            ready=self.ready(),
            resource_count=resource_count,
            informer_count=len(self._async_informers) + len(self._sync_informers),
        )

__init__(**data)

Initialize cache with configuration.

Source code in cloudcoil/caching/_cache.py
46
47
48
49
50
51
52
53
54
55
def __init__(self, **data: Any) -> None:
    """Initialize cache with configuration."""
    super().__init__(**data)
    self._client_factory: Optional[Any] = None  # Will be set to client_for method
    self._async_informers: Dict[str, "AsyncInformer[Any]"] = {}
    self._sync_informers: Dict[str, "SyncInformer[Any]"] = {}
    self._started: bool = False
    self._start_tasks: List[asyncio.Task] = []  # Track informer start tasks
    self._pause_count: int = 0  # For nested pause context managers
    self._original_enabled: Optional[bool] = None  # Store original state

async_start() async

Start all informers asynchronously.

Source code in cloudcoil/caching/_cache.py
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
async def async_start(self) -> None:
    """Start all informers asynchronously."""
    if not self.enabled or self._started:
        return

    self._started = True

    # Pre-create informers for configured resources
    if self.resources:
        logger.debug("Pre-creating informers for %d configured resources", len(self.resources))
        for resource_type in self.resources:
            # This will create the informer if it doesn't exist
            self.get_informer(resource_type, sync=False)

    # Start all async informers and track tasks
    self._start_tasks = []
    if self._async_informers:
        for key, informer in self._async_informers.items():
            if isinstance(informer, AsyncInformer):
                logger.debug("Starting informer for %s", key)
                task = asyncio.create_task(informer._start())
                self._start_tasks.append(task)

        # Wait for all informers to start
        if self._start_tasks:
            await asyncio.gather(*self._start_tasks, return_exceptions=True)

    logger.debug("Cache started with %d async informers", len(self._async_informers))

async_stop() async

Stop all informers asynchronously.

Source code in cloudcoil/caching/_cache.py
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
async def async_stop(self) -> None:
    """Stop all informers asynchronously."""
    if not self._started:
        return

    self._started = False

    # Stop all async informers
    if self._async_informers:
        tasks = []
        for informer in self._async_informers.values():
            if isinstance(informer, AsyncInformer):
                tasks.append(informer._stop())
        if tasks:
            await asyncio.gather(*tasks, return_exceptions=True)

    logger.debug("Cache stopped")

async_wait(timeout=None) async

Wait for cache to be ready asynchronously.

Parameters:

Name Type Description Default
timeout Optional[float]

Maximum time to wait (uses sync_timeout if not provided)

None

Returns:

Type Description
bool

True if ready, False if timeout

Source code in cloudcoil/caching/_cache.py
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
async def async_wait(self, timeout: Optional[float] = None) -> bool:
    """Wait for cache to be ready asynchronously.

    Args:
        timeout: Maximum time to wait (uses sync_timeout if not provided)

    Returns:
        True if ready, False if timeout
    """
    if not self.enabled:
        return True

    # If resources are configured but no informers created yet, that's not ready
    if self.resources and not self._async_informers and not self._sync_informers:
        logger.warning("Cache has configured resources but no informers created yet")
        return False

    # If no informers at all and no resources configured, that's OK
    if not self._async_informers and not self._sync_informers and not self.resources:
        return True

    timeout = self.sync_timeout if timeout is None else timeout

    # First wait for any pending start tasks
    if self._start_tasks:
        try:
            await asyncio.wait_for(
                asyncio.gather(*self._start_tasks, return_exceptions=True),
                timeout=timeout / 2,  # Use half timeout for starting
            )
        except asyncio.TimeoutError:
            logger.warning("Timeout waiting for informers to start")
            return False

    # Then wait for all async informers to sync
    sync_tasks = []
    for informer in self._async_informers.values():
        if isinstance(informer, AsyncInformer):
            sync_tasks.append(informer._wait_for_sync(timeout))

    if sync_tasks:
        results = await asyncio.gather(*sync_tasks, return_exceptions=True)
        return all(r is True for r in results)

    return True

fallback_mode()

Temporarily use fallback mode within this context.

Source code in cloudcoil/caching/_cache.py
391
392
393
394
395
396
397
398
399
@contextmanager
def fallback_mode(self) -> Generator[None, None, None]:
    """Temporarily use fallback mode within this context."""
    original_mode = self.mode
    self.mode = "fallback"
    try:
        yield
    finally:
        self.mode = original_mode

get_informer(resource_type, sync=False, namespace=None)

Get or create informer for a specific resource type.

Parameters:

Name Type Description Default
resource_type Type[T]

The resource type

required
sync bool

Whether to return sync wrapper

False
namespace Optional[str]

Optional namespace scope

None

Returns:

Type Description
Optional[Union[AsyncInformer[T], SyncInformer[T]]]

The informer instance, or None if not available

Source code in cloudcoil/caching/_cache.py
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
def get_informer(
    self,
    resource_type: Type[T],
    sync: bool = False,
    namespace: Optional[str] = None,
) -> Optional[Union["AsyncInformer[T]", "SyncInformer[T]"]]:
    """Get or create informer for a specific resource type.

    Args:
        resource_type: The resource type
        sync: Whether to return sync wrapper
        namespace: Optional namespace scope

    Returns:
        The informer instance, or None if not available
    """
    if not self.enabled or not self._client_factory:
        return None

    # Check if we should cache this resource
    if not self.should_cache(resource_type):
        return None

    # Create cache key
    key = f"{resource_type.__module__}.{resource_type.__name__}"
    if namespace:
        key += f":{namespace}"

    if sync:
        # Get or create sync informer
        if key not in self._sync_informers:
            # Create sync client using factory
            sync_client: APIClient[T] = self._create_client(resource_type, True)
            options = self._create_options(resource_type, namespace)

            sync_informer: SyncInformer[T] = SyncInformer(sync_client, options)
            self._sync_informers[key] = sync_informer  # type: ignore[assignment]

            # Auto-start if cache is running
            if self._started:
                sync_informer._start()  # type: ignore[func-returns-value]
                # Note: Sync informers start synchronously, no task tracking needed

        return self._sync_informers[key]  # type: ignore[return-value]
    else:
        # Get or create async informer
        if key not in self._async_informers:
            # Create async client using factory
            async_client: AsyncAPIClient[T] = self._create_client(resource_type, False)
            options = self._create_options(resource_type, namespace)

            async_informer: AsyncInformer[T] = AsyncInformer(async_client, options)
            self._async_informers[key] = async_informer  # type: ignore[assignment]

            # Auto-start if cache is running
            if self._started:
                task = asyncio.create_task(async_informer._start())
                self._start_tasks.append(task)

                # Also create a task to wait for initial sync
                async def wait_for_sync():
                    try:
                        await task  # Wait for start to complete
                        await async_informer._wait_for_sync(self.sync_timeout)
                    except Exception as e:
                        logger.error("Error starting informer for %s: %s", key, e)

                asyncio.create_task(wait_for_sync())

        return self._async_informers[key]  # type: ignore[return-value]

pause()

Temporarily disable cache within this context.

Supports nested usage with reference counting.

Source code in cloudcoil/caching/_cache.py
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
@contextmanager
def pause(self) -> Generator[None, None, None]:
    """Temporarily disable cache within this context.

    Supports nested usage with reference counting.
    """
    self._pause_count += 1
    if self._pause_count == 1:
        # First pause - save state and disable
        self._original_enabled = self.enabled
        self.enabled = False
    try:
        yield
    finally:
        self._pause_count -= 1
        if self._pause_count == 0:
            # Last pause exiting - restore state
            if self._original_enabled is not None:
                self.enabled = self._original_enabled
            self._original_enabled = None

ready()

Check if cache is synced and ready (non-blocking check).

Source code in cloudcoil/caching/_cache.py
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
def ready(self) -> bool:
    """Check if cache is synced and ready (non-blocking check)."""
    if not self.enabled:
        return True
    if not self._started:
        return False

    # Check if all informers have synced
    for informer in self._async_informers.values():
        if isinstance(informer, AsyncInformer) and not informer._has_synced():
            return False

    for sync_informer in self._sync_informers.values():
        if isinstance(sync_informer, SyncInformer) and not sync_informer._has_synced():
            return False

    return True

set_client_factory(factory)

Set the client factory function (called by Config).

Source code in cloudcoil/caching/_cache.py
57
58
59
def set_client_factory(self, factory: Any) -> None:
    """Set the client factory function (called by Config)."""
    self._client_factory = factory

start()

Start all informers synchronously (blocks until started).

Source code in cloudcoil/caching/_cache.py
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
def start(self) -> None:
    """Start all informers synchronously (blocks until started)."""
    if not self.enabled or self._started:
        return

    self._started = True

    # Pre-create informers for configured resources
    if self.resources:
        logger.debug(
            "Pre-creating sync informers for %d configured resources", len(self.resources)
        )
        for resource_type in self.resources:
            # This will create the informer if it doesn't exist
            self.get_informer(resource_type, sync=True)

    # Start all sync informers
    for informer in self._sync_informers.values():
        if isinstance(informer, SyncInformer):
            informer._start()

    logger.debug("Cache started with %d sync informers", len(self._sync_informers))

status()

Get cache status information.

Source code in cloudcoil/caching/_cache.py
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
def status(self) -> CacheStatus:
    """Get cache status information."""
    if not self.enabled:
        return CacheStatus(enabled=False)

    if not self._client_factory:
        return CacheStatus(enabled=True, ready=False)

    # Count total resources across all informers
    resource_count = 0
    for informer in self._async_informers.values():
        resource_count += informer._store.size()
    for sync_informer in self._sync_informers.values():
        resource_count += sync_informer._store.size()

    return CacheStatus(
        enabled=True,
        started=self._started,
        ready=self.ready(),
        resource_count=resource_count,
        informer_count=len(self._async_informers) + len(self._sync_informers),
    )

stop()

Stop all informers synchronously (blocks until stopped).

Source code in cloudcoil/caching/_cache.py
196
197
198
199
200
201
202
203
204
205
206
207
208
def stop(self) -> None:
    """Stop all informers synchronously (blocks until stopped)."""
    if not self._started:
        return

    self._started = False

    # Stop all sync informers
    for informer in self._sync_informers.values():
        if isinstance(informer, SyncInformer):
            informer._stop()

    logger.debug("Cache stopped")

strict_mode()

Temporarily use strict mode within this context.

Source code in cloudcoil/caching/_cache.py
381
382
383
384
385
386
387
388
389
@contextmanager
def strict_mode(self) -> Generator[None, None, None]:
    """Temporarily use strict mode within this context."""
    original_mode = self.mode
    self.mode = "strict"
    try:
        yield
    finally:
        self.mode = original_mode

wait(timeout=None)

Wait for cache to be ready synchronously.

Parameters:

Name Type Description Default
timeout Optional[float]

Maximum time to wait (uses sync_timeout if not provided)

None

Returns:

Type Description
bool

True if ready, False if timeout

Source code in cloudcoil/caching/_cache.py
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
def wait(self, timeout: Optional[float] = None) -> bool:
    """Wait for cache to be ready synchronously.

    Args:
        timeout: Maximum time to wait (uses sync_timeout if not provided)

    Returns:
        True if ready, False if timeout
    """
    if not self.enabled:
        return True

    # If resources are configured but no informers created yet, that's not ready
    if self.resources and not self._sync_informers and not self._async_informers:
        logger.warning("Cache has configured resources but no informers created yet")
        return False

    # If no informers at all and no resources configured, that's OK
    if not self._sync_informers and not self._async_informers and not self.resources:
        return True

    timeout = self.sync_timeout if timeout is None else timeout

    # Wait for all sync informers to sync
    start_time = time.time()
    for informer in self._sync_informers.values():
        if isinstance(informer, SyncInformer):
            remaining = max(0, timeout - (time.time() - start_time))
            if not informer._wait_for_sync(remaining):
                return False

    return True

CachedResources

Local get/list with no API fallback; values are independent deep copies.

Missing objects return None. An unsynced or stopped informer raises instead of presenting an empty cache as authoritative. Lists cover only the watched scope, are eventually consistent, and have no server pagination tokens.

Source code in cloudcoil/caching/_reader.py
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
class CachedResources[T: Resource]:
    """Local get/list with no API fallback; values are independent deep copies.

    Missing objects return None. An unsynced or stopped informer raises instead
    of presenting an empty cache as authoritative. Lists cover only the watched
    scope, are eventually consistent, and have no server pagination tokens.
    """

    def __init__(self, informer: AsyncInformer[T], namespace: str | None = None) -> None:
        self._informer = informer
        self._namespace = namespace

    def _check(self) -> None:
        if not self._informer._has_synced():
            raise RuntimeError("Informer is not running and synced")
        if self._informer._watch._error is not None:
            raise RuntimeError("Informer watch failed") from self._informer._watch._error

    def get(self, name: str, namespace: str | None = None) -> T | None:
        self._check()
        namespace = (namespace or self._namespace) if self._informer._client.namespaced else None
        if self._informer._client.namespaced and namespace is None:
            raise ValueError("Specify a namespace for a namespaced cache lookup")
        obj = self._informer.get(name, namespace)
        return obj.model_copy(deep=True) if obj is not None else None

    def list(
        self,
        namespace: str | None = None,
        *,
        all_namespaces: bool = False,
        labels: Mapping[str, str] | None = None,
    ) -> list[T]:
        """List cached objects; labels is an AND of exact label matches.

        all_namespaces means all namespaces in this informer's configured scope,
        not an expansion of its watch. Filtering scans the in-memory snapshot.
        """
        self._check()
        if all_namespaces and namespace is not None:
            raise ValueError("namespace and all_namespaces are mutually exclusive")
        target = (
            None
            if all_namespaces or not self._informer._client.namespaced
            else namespace or self._namespace
        )
        if self._informer._client.namespaced and target is None and not all_namespaces:
            raise ValueError("Specify namespace or all_namespaces=True")
        return [
            obj.model_copy(deep=True)
            for obj in self._informer.list(namespace=target)
            if not labels
            or all(
                obj.metadata and (obj.metadata.labels or {}).get(key) == value
                for key, value in labels.items()
            )
        ]

list(namespace=None, *, all_namespaces=False, labels=None)

List cached objects; labels is an AND of exact label matches.

all_namespaces means all namespaces in this informer's configured scope, not an expansion of its watch. Filtering scans the in-memory snapshot.

Source code in cloudcoil/caching/_reader.py
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
def list(
    self,
    namespace: str | None = None,
    *,
    all_namespaces: bool = False,
    labels: Mapping[str, str] | None = None,
) -> list[T]:
    """List cached objects; labels is an AND of exact label matches.

    all_namespaces means all namespaces in this informer's configured scope,
    not an expansion of its watch. Filtering scans the in-memory snapshot.
    """
    self._check()
    if all_namespaces and namespace is not None:
        raise ValueError("namespace and all_namespaces are mutually exclusive")
    target = (
        None
        if all_namespaces or not self._informer._client.namespaced
        else namespace or self._namespace
    )
    if self._informer._client.namespaced and target is None and not all_namespaces:
        raise ValueError("Specify namespace or all_namespaces=True")
    return [
        obj.model_copy(deep=True)
        for obj in self._informer.list(namespace=target)
        if not labels
        or all(
            obj.metadata and (obj.metadata.labels or {}).get(key) == value
            for key, value in labels.items()
        )
    ]

AsyncInformer

Bases: Generic[T]

Async informer with minimal public API and better concurrency patterns.

Public API: - get(name, namespace) - Get resource from cache - list(namespace, label_selector, field_selector) - List resources from cache - on_add(handler) - Register add event handler - on_update(handler) - Register update event handler - on_delete(handler) - Register delete event handler

All other methods and attributes are private implementation details.

Source code in cloudcoil/caching/_informer.py
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
class AsyncInformer(Generic[T]):
    """Async informer with minimal public API and better concurrency patterns.

    Public API:
    - get(name, namespace) - Get resource from cache
    - list(namespace, label_selector, field_selector) - List resources from cache
    - on_add(handler) - Register add event handler
    - on_update(handler) - Register update event handler
    - on_delete(handler) - Register delete event handler

    All other methods and attributes are private implementation details.
    """

    def __init__(
        self,
        client: AsyncAPIClient[T],
        options: InformerOptions,
    ):
        """Initialize the async informer."""
        self._client = client
        self._options = options

        # Components
        self._store = ConcurrentStore[T](max_items=options.max_items)
        self._dispatcher = _AsyncEventDispatcher[T]()
        self._watch = _AsyncWatchManager(
            client=client,
            options=options,
            on_items_callback=self._handle_initial_items,
            on_event_callback=self._handle_watch_event,
        )

        # Sync state
        self._sync_event = asyncio.Event()
        self._started = False

        # Pending handlers to register when loop is available
        self._pending_handlers: List[tuple] = []

    # ============ Public Read API ============

    def get(self, name: str, namespace: Optional[str] = None) -> Optional[T]:
        """Get a resource from cache by name."""
        return self._store.get_by_name(name, namespace)

    def list(
        self,
        namespace: Optional[str] = None,
        label_selector: Optional[str] = None,
        field_selector: Optional[str] = None,
    ) -> List[T]:
        """List resources from cache with optional filtering."""
        items = self._store.list()

        # Apply filters
        if namespace is not None:
            items = [
                item
                for item in items
                if hasattr(item, "metadata")
                and item.metadata
                and item.metadata.namespace == namespace
            ]

        if label_selector:
            items = self._filter_by_labels(items, label_selector)

        if field_selector:
            items = self._filter_by_fields(items, field_selector)

        return items

    # ============ Public Event API ============

    @overload
    def on_add(self, handler: Union[EventHandler, AsyncEventHandler]) -> None:
        """Register a handler for add events."""
        ...

    @overload
    def on_add(
        self, handler: None = None
    ) -> Callable[[Union[EventHandler, AsyncEventHandler]], Union[EventHandler, AsyncEventHandler]]:
        """Use as a decorator to register a handler for add events."""
        ...

    def on_add(self, handler: Optional[Union[EventHandler, AsyncEventHandler]] = None) -> Any:
        """Register a handler for add events. Can be used as a decorator.

        Usage:
            # As a decorator
            @informer.on_add
            def handle_add(obj):
                print(f"Added: {obj.metadata.name}")

            # As a regular method
            informer.on_add(handle_add)
        """

        def register(
            h: Union[EventHandler, AsyncEventHandler],
        ) -> Union[EventHandler, AsyncEventHandler]:
            try:
                # Try to get the running loop
                loop = asyncio.get_running_loop()
                # If we have a loop, create the task
                if self._started:
                    loop.create_task(self._dispatcher.register_add_handler(h))
                else:
                    self._pending_handlers.append(("add", h))
            except RuntimeError:
                # No running loop - store for later registration
                self._pending_handlers.append(("add", h))
            return h

        if handler is None:
            # Being used as a decorator
            return register
        else:
            # Being called directly with a handler
            register(handler)
            return None

    @overload
    def on_update(self, handler: Union[UpdateHandler, AsyncUpdateHandler]) -> None:
        """Register a handler for update events."""
        ...

    @overload
    def on_update(
        self, handler: None = None
    ) -> Callable[
        [Union[UpdateHandler, AsyncUpdateHandler]], Union[UpdateHandler, AsyncUpdateHandler]
    ]:
        """Use as a decorator to register a handler for update events."""
        ...

    def on_update(self, handler: Optional[Union[UpdateHandler, AsyncUpdateHandler]] = None) -> Any:
        """Register a handler for update events. Can be used as a decorator.

        Usage:
            # As a decorator
            @informer.on_update
            def handle_update(old_obj, new_obj):
                print(f"Updated: {new_obj.metadata.name}")

            # As a regular method
            informer.on_update(handle_update)
        """

        def register(
            h: Union[UpdateHandler, AsyncUpdateHandler],
        ) -> Union[UpdateHandler, AsyncUpdateHandler]:
            try:
                loop = asyncio.get_running_loop()
                if self._started:
                    loop.create_task(self._dispatcher.register_update_handler(h))
                else:
                    self._pending_handlers.append(("update", h))
            except RuntimeError:
                # No running loop - store for later registration
                self._pending_handlers.append(("update", h))
            return h

        if handler is None:
            # Being used as a decorator
            return register
        else:
            # Being called directly with a handler
            register(handler)
            return None

    @overload
    def on_delete(self, handler: Union[EventHandler, AsyncEventHandler]) -> None:
        """Register a handler for delete events."""
        ...

    @overload
    def on_delete(
        self, handler: None = None
    ) -> Callable[[Union[EventHandler, AsyncEventHandler]], Union[EventHandler, AsyncEventHandler]]:
        """Use as a decorator to register a handler for delete events."""
        ...

    def on_delete(self, handler: Optional[Union[EventHandler, AsyncEventHandler]] = None) -> Any:
        """Register a handler for delete events. Can be used as a decorator.

        Usage:
            # As a decorator
            @informer.on_delete
            def handle_delete(obj):
                print(f"Deleted: {obj.metadata.name}")

            # As a regular method
            informer.on_delete(handle_delete)
        """

        def register(
            h: Union[EventHandler, AsyncEventHandler],
        ) -> Union[EventHandler, AsyncEventHandler]:
            try:
                loop = asyncio.get_running_loop()
                if self._started:
                    loop.create_task(self._dispatcher.register_delete_handler(h))
                else:
                    self._pending_handlers.append(("delete", h))
            except RuntimeError:
                # No running loop - store for later registration
                self._pending_handlers.append(("delete", h))
            return h

        if handler is None:
            # Being used as a decorator
            return register
        else:
            # Being called directly with a handler
            register(handler)
            return None

    # ============ Public Index API (Advanced Usage) ============

    def add_index(self, name: str, index_func: Callable[[T], str]) -> None:
        """Add a custom index for fast lookups.

        Args:
            name: The name of the index
            index_func: Function that extracts index key from a resource
        """
        self._store.add_index(name, index_func)

    def get_by_index(self, index_name: str, index_key: str) -> List[T]:
        """Get resources by custom index."""
        return self._store.get_by_index(index_name, index_key)

    def list_index_keys(self, index_name: str) -> List[str]:
        """List all keys for a given index."""
        return self._store.list_index_keys(index_name)

    # ============ Public Status API ============

    def has_synced(self) -> bool:
        """Check if initial sync is complete."""
        return self._has_synced()

    # ============ Private Implementation ============

    async def _start(self) -> None:
        """Start the informer (internal use only)."""
        if self._started:
            return
        self._started = True
        self._sync_event.clear()

        # Register any pending handlers now that we have a loop
        for handler_type, handler in self._pending_handlers:
            if handler_type == "add":
                await self._dispatcher.register_add_handler(handler)
            elif handler_type == "update":
                await self._dispatcher.register_update_handler(handler)
            elif handler_type == "delete":
                await self._dispatcher.register_delete_handler(handler)
        self._pending_handlers.clear()

        await self._watch.start()

    async def _stop(self) -> None:
        """Stop the informer (internal use only)."""
        if not self._started:
            return
        self._started = False
        await self._watch.stop()

    async def _wait_for_sync(self, timeout: float = 30.0) -> bool:
        """Wait for initial sync (internal use only)."""
        try:
            await asyncio.wait_for(self._sync_event.wait(), timeout=timeout)
            return True
        except asyncio.TimeoutError:
            return False

    def _has_synced(self) -> bool:
        """Check if initial sync complete (internal use only)."""
        # The initial-list callback wakes waiters before the watch loop changes state.
        # Readiness describes the populated snapshot, not the transport's state.
        return self._sync_event.is_set() and self._started

    async def _handle_initial_items(self, items: List[T], resource_version: str) -> None:
        """Handle initial list of items."""
        previous = {self._get_key(obj): obj for obj in self._store.list()}
        current = {self._get_key(obj): obj for obj in items}
        await self._store.async_replace(items)
        for key, old in previous.items():
            new = current.get(key)
            if new is None or _uid(old) != _uid(new):
                await self._dispatcher.dispatch_delete(old)
        for key, new in current.items():
            prior = previous.get(key)
            if prior is None or _uid(prior) != _uid(new):
                await self._dispatcher.dispatch_add(new)
            else:
                # Relists also resync unchanged objects to repair external drift.
                await self._dispatcher.dispatch_update(prior, new)
        self._sync_event.set()
        logger.info(
            "Initial sync complete for %s: %d items", self._client.kind.__name__, len(items)
        )

    async def _handle_watch_event(self, event_type: str, obj: Any) -> None:
        """Handle watch event."""
        if event_type == "BOOKMARK":
            # Just a resource version update
            return
        elif event_type == "ADDED":
            await self._handle_add(obj)
        elif event_type == "MODIFIED":
            await self._handle_update(obj)
        elif event_type == "DELETED":
            await self._handle_delete(obj)
        elif event_type == "ERROR":
            logger.error("Watch error event: %s", obj)

    async def _handle_add(self, obj: T) -> None:
        """Handle resource add."""
        await self._store.async_add(obj)
        await self._dispatcher.dispatch_add(obj)

    async def _handle_update(self, obj: T) -> None:
        """Handle resource update."""
        key = self._get_key(obj)
        old_obj = self._store.get(key) if key else None
        await self._store.async_add(obj)  # add acts as update
        await self._dispatcher.dispatch_update(old_obj, obj)

    async def _handle_delete(self, obj: T) -> None:
        """Handle resource delete."""
        await self._store.async_delete(obj)
        await self._dispatcher.dispatch_delete(obj)

    def _get_key(self, obj: T) -> Optional[str]:
        """Get cache key for object."""
        if hasattr(obj, "metadata") and obj.metadata:
            namespace = obj.metadata.namespace
            name = obj.metadata.name
            return f"{namespace}/{name}" if namespace and name else name
        return None

    def _filter_by_labels(self, items: List[T], selector: str) -> List[T]:
        """Filter items by label selector."""
        # Simple implementation - just check for exact matches
        parts = selector.split("=")
        if len(parts) != 2:
            return items

        key, value = parts[0].strip(), parts[1].strip()
        return [
            item
            for item in items
            if hasattr(item, "metadata")
            and item.metadata
            and item.metadata.labels
            and item.metadata.labels.get(key) == value
        ]

    def _filter_by_fields(self, items: List[T], selector: str) -> List[T]:
        """Filter items by field selector."""
        # Simple implementation for metadata.name and metadata.namespace
        parts = selector.split("=")
        if len(parts) != 2:
            return items

        field, value = parts[0].strip(), parts[1].strip()

        if field == "metadata.name":
            return [
                item
                for item in items
                if hasattr(item, "metadata") and item.metadata and item.metadata.name == value
            ]
        elif field == "metadata.namespace":
            return [
                item
                for item in items
                if hasattr(item, "metadata") and item.metadata and item.metadata.namespace == value
            ]

        return items

__init__(client, options)

Initialize the async informer.

Source code in cloudcoil/caching/_informer.py
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
def __init__(
    self,
    client: AsyncAPIClient[T],
    options: InformerOptions,
):
    """Initialize the async informer."""
    self._client = client
    self._options = options

    # Components
    self._store = ConcurrentStore[T](max_items=options.max_items)
    self._dispatcher = _AsyncEventDispatcher[T]()
    self._watch = _AsyncWatchManager(
        client=client,
        options=options,
        on_items_callback=self._handle_initial_items,
        on_event_callback=self._handle_watch_event,
    )

    # Sync state
    self._sync_event = asyncio.Event()
    self._started = False

    # Pending handlers to register when loop is available
    self._pending_handlers: List[tuple] = []

add_index(name, index_func)

Add a custom index for fast lookups.

Parameters:

Name Type Description Default
name str

The name of the index

required
index_func Callable[[T], str]

Function that extracts index key from a resource

required
Source code in cloudcoil/caching/_informer.py
252
253
254
255
256
257
258
259
def add_index(self, name: str, index_func: Callable[[T], str]) -> None:
    """Add a custom index for fast lookups.

    Args:
        name: The name of the index
        index_func: Function that extracts index key from a resource
    """
    self._store.add_index(name, index_func)

get(name, namespace=None)

Get a resource from cache by name.

Source code in cloudcoil/caching/_informer.py
72
73
74
def get(self, name: str, namespace: Optional[str] = None) -> Optional[T]:
    """Get a resource from cache by name."""
    return self._store.get_by_name(name, namespace)

get_by_index(index_name, index_key)

Get resources by custom index.

Source code in cloudcoil/caching/_informer.py
261
262
263
def get_by_index(self, index_name: str, index_key: str) -> List[T]:
    """Get resources by custom index."""
    return self._store.get_by_index(index_name, index_key)

has_synced()

Check if initial sync is complete.

Source code in cloudcoil/caching/_informer.py
271
272
273
def has_synced(self) -> bool:
    """Check if initial sync is complete."""
    return self._has_synced()

list(namespace=None, label_selector=None, field_selector=None)

List resources from cache with optional filtering.

Source code in cloudcoil/caching/_informer.py
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
def list(
    self,
    namespace: Optional[str] = None,
    label_selector: Optional[str] = None,
    field_selector: Optional[str] = None,
) -> List[T]:
    """List resources from cache with optional filtering."""
    items = self._store.list()

    # Apply filters
    if namespace is not None:
        items = [
            item
            for item in items
            if hasattr(item, "metadata")
            and item.metadata
            and item.metadata.namespace == namespace
        ]

    if label_selector:
        items = self._filter_by_labels(items, label_selector)

    if field_selector:
        items = self._filter_by_fields(items, field_selector)

    return items

list_index_keys(index_name)

List all keys for a given index.

Source code in cloudcoil/caching/_informer.py
265
266
267
def list_index_keys(self, index_name: str) -> List[str]:
    """List all keys for a given index."""
    return self._store.list_index_keys(index_name)

on_add(handler=None)

on_add(
    handler: Union[EventHandler, AsyncEventHandler],
) -> None
on_add(
    handler: None = None,
) -> Callable[
    [Union[EventHandler, AsyncEventHandler]],
    Union[EventHandler, AsyncEventHandler],
]

Register a handler for add events. Can be used as a decorator.

Usage

As a decorator

@informer.on_add def handle_add(obj): print(f"Added: {obj.metadata.name}")

As a regular method

informer.on_add(handle_add)

Source code in cloudcoil/caching/_informer.py
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
def on_add(self, handler: Optional[Union[EventHandler, AsyncEventHandler]] = None) -> Any:
    """Register a handler for add events. Can be used as a decorator.

    Usage:
        # As a decorator
        @informer.on_add
        def handle_add(obj):
            print(f"Added: {obj.metadata.name}")

        # As a regular method
        informer.on_add(handle_add)
    """

    def register(
        h: Union[EventHandler, AsyncEventHandler],
    ) -> Union[EventHandler, AsyncEventHandler]:
        try:
            # Try to get the running loop
            loop = asyncio.get_running_loop()
            # If we have a loop, create the task
            if self._started:
                loop.create_task(self._dispatcher.register_add_handler(h))
            else:
                self._pending_handlers.append(("add", h))
        except RuntimeError:
            # No running loop - store for later registration
            self._pending_handlers.append(("add", h))
        return h

    if handler is None:
        # Being used as a decorator
        return register
    else:
        # Being called directly with a handler
        register(handler)
        return None

on_delete(handler=None)

on_delete(
    handler: Union[EventHandler, AsyncEventHandler],
) -> None
on_delete(
    handler: None = None,
) -> Callable[
    [Union[EventHandler, AsyncEventHandler]],
    Union[EventHandler, AsyncEventHandler],
]

Register a handler for delete events. Can be used as a decorator.

Usage

As a decorator

@informer.on_delete def handle_delete(obj): print(f"Deleted: {obj.metadata.name}")

As a regular method

informer.on_delete(handle_delete)

Source code in cloudcoil/caching/_informer.py
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
def on_delete(self, handler: Optional[Union[EventHandler, AsyncEventHandler]] = None) -> Any:
    """Register a handler for delete events. Can be used as a decorator.

    Usage:
        # As a decorator
        @informer.on_delete
        def handle_delete(obj):
            print(f"Deleted: {obj.metadata.name}")

        # As a regular method
        informer.on_delete(handle_delete)
    """

    def register(
        h: Union[EventHandler, AsyncEventHandler],
    ) -> Union[EventHandler, AsyncEventHandler]:
        try:
            loop = asyncio.get_running_loop()
            if self._started:
                loop.create_task(self._dispatcher.register_delete_handler(h))
            else:
                self._pending_handlers.append(("delete", h))
        except RuntimeError:
            # No running loop - store for later registration
            self._pending_handlers.append(("delete", h))
        return h

    if handler is None:
        # Being used as a decorator
        return register
    else:
        # Being called directly with a handler
        register(handler)
        return None

on_update(handler=None)

on_update(
    handler: Union[UpdateHandler, AsyncUpdateHandler],
) -> None
on_update(
    handler: None = None,
) -> Callable[
    [Union[UpdateHandler, AsyncUpdateHandler]],
    Union[UpdateHandler, AsyncUpdateHandler],
]

Register a handler for update events. Can be used as a decorator.

Usage

As a decorator

@informer.on_update def handle_update(old_obj, new_obj): print(f"Updated: {new_obj.metadata.name}")

As a regular method

informer.on_update(handle_update)

Source code in cloudcoil/caching/_informer.py
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
def on_update(self, handler: Optional[Union[UpdateHandler, AsyncUpdateHandler]] = None) -> Any:
    """Register a handler for update events. Can be used as a decorator.

    Usage:
        # As a decorator
        @informer.on_update
        def handle_update(old_obj, new_obj):
            print(f"Updated: {new_obj.metadata.name}")

        # As a regular method
        informer.on_update(handle_update)
    """

    def register(
        h: Union[UpdateHandler, AsyncUpdateHandler],
    ) -> Union[UpdateHandler, AsyncUpdateHandler]:
        try:
            loop = asyncio.get_running_loop()
            if self._started:
                loop.create_task(self._dispatcher.register_update_handler(h))
            else:
                self._pending_handlers.append(("update", h))
        except RuntimeError:
            # No running loop - store for later registration
            self._pending_handlers.append(("update", h))
        return h

    if handler is None:
        # Being used as a decorator
        return register
    else:
        # Being called directly with a handler
        register(handler)
        return None

SyncInformer

Bases: Generic[T]

Sync informer with minimal public API and better threading patterns.

Public API: - get(name, namespace) - Get resource from cache - list(namespace, label_selector, field_selector) - List resources from cache - on_add(handler) - Register add event handler - on_update(handler) - Register update event handler - on_delete(handler) - Register delete event handler

All other methods and attributes are private implementation details.

Source code in cloudcoil/caching/_informer.py
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
class SyncInformer(Generic[T]):
    """Sync informer with minimal public API and better threading patterns.

    Public API:
    - get(name, namespace) - Get resource from cache
    - list(namespace, label_selector, field_selector) - List resources from cache
    - on_add(handler) - Register add event handler
    - on_update(handler) - Register update event handler
    - on_delete(handler) - Register delete event handler

    All other methods and attributes are private implementation details.
    """

    def __init__(
        self,
        client: APIClient[T],
        options: InformerOptions,
    ):
        """Initialize the sync informer."""
        self._client = client
        self._options = options

        # Components
        self._store = ConcurrentStore[T](max_items=options.max_items)
        self._dispatcher = _SyncEventDispatcher[T]()
        self._watch = _SyncWatchManager(
            client=client,
            options=options,
            on_items_callback=self._handle_initial_items,
            on_event_callback=self._handle_watch_event,
        )

        # Sync state
        self._sync_event = threading.Event()
        self._started = False
        self._started_lock = threading.Lock()

    # ============ Public Read API ============

    def get(self, name: str, namespace: Optional[str] = None) -> Optional[T]:
        """Get a resource from cache by name."""
        return self._store.get_by_name(name, namespace)

    def list(
        self,
        namespace: Optional[str] = None,
        label_selector: Optional[str] = None,
        field_selector: Optional[str] = None,
    ) -> List[T]:
        """List resources from cache with optional filtering."""
        items = self._store.list()

        # Apply filters
        if namespace is not None:
            items = [
                item
                for item in items
                if hasattr(item, "metadata")
                and item.metadata
                and item.metadata.namespace == namespace
            ]

        if label_selector:
            items = self._filter_by_labels(items, label_selector)

        if field_selector:
            items = self._filter_by_fields(items, field_selector)

        return items

    # ============ Public Event API ============

    @overload
    def on_add(self, handler: EventHandler) -> None:
        """Register a handler for add events."""
        ...

    @overload
    def on_add(self, handler: None = None) -> Callable[[EventHandler], EventHandler]:
        """Use as a decorator to register a handler for add events."""
        ...

    def on_add(self, handler: Optional[EventHandler] = None) -> Any:
        """Register a handler for add events. Can be used as a decorator.

        Usage:
            # As a decorator
            @informer.on_add
            def handle_add(obj):
                print(f"Added: {obj.metadata.name}")

            # As a regular method
            informer.on_add(handle_add)
        """

        def register(h: EventHandler) -> EventHandler:
            self._dispatcher.register_add_handler(h)
            return h

        if handler is None:
            # Being used as a decorator
            return register
        else:
            # Being called directly with a handler
            register(handler)
            return None

    @overload
    def on_update(self, handler: UpdateHandler) -> None:
        """Register a handler for update events."""
        ...

    @overload
    def on_update(self, handler: None = None) -> Callable[[UpdateHandler], UpdateHandler]:
        """Use as a decorator to register a handler for update events."""
        ...

    def on_update(self, handler: Optional[UpdateHandler] = None) -> Any:
        """Register a handler for update events. Can be used as a decorator.

        Usage:
            # As a decorator
            @informer.on_update
            def handle_update(old_obj, new_obj):
                print(f"Updated: {new_obj.metadata.name}")

            # As a regular method
            informer.on_update(handle_update)
        """

        def register(h: UpdateHandler) -> UpdateHandler:
            self._dispatcher.register_update_handler(h)
            return h

        if handler is None:
            # Being used as a decorator
            return register
        else:
            # Being called directly with a handler
            register(handler)
            return None

    @overload
    def on_delete(self, handler: EventHandler) -> None:
        """Register a handler for delete events."""
        ...

    @overload
    def on_delete(self, handler: None = None) -> Callable[[EventHandler], EventHandler]:
        """Use as a decorator to register a handler for delete events."""
        ...

    def on_delete(self, handler: Optional[EventHandler] = None) -> Any:
        """Register a handler for delete events. Can be used as a decorator.

        Usage:
            # As a decorator
            @informer.on_delete
            def handle_delete(obj):
                print(f"Deleted: {obj.metadata.name}")

            # As a regular method
            informer.on_delete(handle_delete)
        """

        def register(h: EventHandler) -> EventHandler:
            self._dispatcher.register_delete_handler(h)
            return h

        if handler is None:
            # Being used as a decorator
            return register
        else:
            # Being called directly with a handler
            register(handler)
            return None

    # ============ Public Index API (Advanced Usage) ============

    def add_index(self, name: str, index_func: Callable[[T], str]) -> None:
        """Add a custom index for fast lookups.

        Args:
            name: The name of the index
            index_func: Function that extracts index key from a resource
        """
        self._store.add_index(name, index_func)

    def get_by_index(self, index_name: str, index_key: str) -> List[T]:
        """Get resources by custom index."""
        return self._store.get_by_index(index_name, index_key)

    def list_index_keys(self, index_name: str) -> List[str]:
        """List all keys for a given index."""
        return self._store.list_index_keys(index_name)

    # ============ Public Status API ============

    def has_synced(self) -> bool:
        """Check if initial sync is complete."""
        return self._has_synced()

    # ============ Private Implementation ============

    def _start(self) -> None:
        """Start the informer (internal use only)."""
        with self._started_lock:
            if self._started:
                return
            self._started = True
            self._sync_event.clear()
            self._watch.start()

    def _stop(self) -> None:
        """Stop the informer (internal use only)."""
        with self._started_lock:
            if not self._started:
                return
            self._started = False
            self._watch.stop()

    def _wait_for_sync(self, timeout: float = 30.0) -> bool:
        """Wait for initial sync (internal use only)."""
        return self._sync_event.wait(timeout=timeout)

    def _has_synced(self) -> bool:
        """Check if initial sync complete (internal use only)."""
        # The initial-list callback wakes waiters before the watch loop changes state.
        # Readiness describes the populated snapshot, not the transport's state.
        return self._sync_event.is_set() and self._started

    def _handle_initial_items(self, items: List[T], resource_version: str) -> None:
        """Handle initial list of items."""
        previous = {self._get_key(obj): obj for obj in self._store.list()}
        current = {self._get_key(obj): obj for obj in items}
        self._store.replace(items)
        for key, old in previous.items():
            new = current.get(key)
            if new is None or _uid(old) != _uid(new):
                self._dispatcher.dispatch_delete(old)
        for key, new in current.items():
            prior = previous.get(key)
            if prior is None or _uid(prior) != _uid(new):
                self._dispatcher.dispatch_add(new)
            else:
                self._dispatcher.dispatch_update(prior, new)
        self._sync_event.set()
        logger.info(
            "Initial sync complete for %s: %d items", self._client.kind.__name__, len(items)
        )

    def _handle_watch_event(self, event_type: str, obj: Any) -> None:
        """Handle watch event."""
        if event_type == "BOOKMARK":
            return
        elif event_type == "ADDED":
            self._handle_add(obj)
        elif event_type == "MODIFIED":
            self._handle_update(obj)
        elif event_type == "DELETED":
            self._handle_delete(obj)
        elif event_type == "ERROR":
            logger.error("Watch error event: %s", obj)

    def _handle_add(self, obj: T) -> None:
        """Handle resource add."""
        self._store.add(obj)
        self._dispatcher.dispatch_add(obj)

    def _handle_update(self, obj: T) -> None:
        """Handle resource update."""
        key = self._get_key(obj)
        old_obj = self._store.get(key) if key else None
        self._store.add(obj)  # add works as update too
        self._dispatcher.dispatch_update(old_obj, obj)

    def _handle_delete(self, obj: T) -> None:
        """Handle resource delete."""
        self._store.delete(obj)
        self._dispatcher.dispatch_delete(obj)

    def _get_key(self, obj: T) -> Optional[str]:
        """Get cache key for object."""
        if hasattr(obj, "metadata") and obj.metadata:
            namespace = obj.metadata.namespace
            name = obj.metadata.name
            return f"{namespace}/{name}" if namespace and name else name
        return None

    def _filter_by_labels(self, items: List[T], selector: str) -> List[T]:
        """Filter items by label selector."""
        parts = selector.split("=")
        if len(parts) != 2:
            return items

        key, value = parts[0].strip(), parts[1].strip()
        return [
            item
            for item in items
            if hasattr(item, "metadata")
            and item.metadata
            and item.metadata.labels
            and item.metadata.labels.get(key) == value
        ]

    def _filter_by_fields(self, items: List[T], selector: str) -> List[T]:
        """Filter items by field selector."""
        parts = selector.split("=")
        if len(parts) != 2:
            return items

        field, value = parts[0].strip(), parts[1].strip()

        if field == "metadata.name":
            return [
                item
                for item in items
                if hasattr(item, "metadata") and item.metadata and item.metadata.name == value
            ]
        elif field == "metadata.namespace":
            return [
                item
                for item in items
                if hasattr(item, "metadata") and item.metadata and item.metadata.namespace == value
            ]

        return items

__init__(client, options)

Initialize the sync informer.

Source code in cloudcoil/caching/_informer.py
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
def __init__(
    self,
    client: APIClient[T],
    options: InformerOptions,
):
    """Initialize the sync informer."""
    self._client = client
    self._options = options

    # Components
    self._store = ConcurrentStore[T](max_items=options.max_items)
    self._dispatcher = _SyncEventDispatcher[T]()
    self._watch = _SyncWatchManager(
        client=client,
        options=options,
        on_items_callback=self._handle_initial_items,
        on_event_callback=self._handle_watch_event,
    )

    # Sync state
    self._sync_event = threading.Event()
    self._started = False
    self._started_lock = threading.Lock()

add_index(name, index_func)

Add a custom index for fast lookups.

Parameters:

Name Type Description Default
name str

The name of the index

required
index_func Callable[[T], str]

Function that extracts index key from a resource

required
Source code in cloudcoil/caching/_informer.py
598
599
600
601
602
603
604
605
def add_index(self, name: str, index_func: Callable[[T], str]) -> None:
    """Add a custom index for fast lookups.

    Args:
        name: The name of the index
        index_func: Function that extracts index key from a resource
    """
    self._store.add_index(name, index_func)

get(name, namespace=None)

Get a resource from cache by name.

Source code in cloudcoil/caching/_informer.py
458
459
460
def get(self, name: str, namespace: Optional[str] = None) -> Optional[T]:
    """Get a resource from cache by name."""
    return self._store.get_by_name(name, namespace)

get_by_index(index_name, index_key)

Get resources by custom index.

Source code in cloudcoil/caching/_informer.py
607
608
609
def get_by_index(self, index_name: str, index_key: str) -> List[T]:
    """Get resources by custom index."""
    return self._store.get_by_index(index_name, index_key)

has_synced()

Check if initial sync is complete.

Source code in cloudcoil/caching/_informer.py
617
618
619
def has_synced(self) -> bool:
    """Check if initial sync is complete."""
    return self._has_synced()

list(namespace=None, label_selector=None, field_selector=None)

List resources from cache with optional filtering.

Source code in cloudcoil/caching/_informer.py
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
def list(
    self,
    namespace: Optional[str] = None,
    label_selector: Optional[str] = None,
    field_selector: Optional[str] = None,
) -> List[T]:
    """List resources from cache with optional filtering."""
    items = self._store.list()

    # Apply filters
    if namespace is not None:
        items = [
            item
            for item in items
            if hasattr(item, "metadata")
            and item.metadata
            and item.metadata.namespace == namespace
        ]

    if label_selector:
        items = self._filter_by_labels(items, label_selector)

    if field_selector:
        items = self._filter_by_fields(items, field_selector)

    return items

list_index_keys(index_name)

List all keys for a given index.

Source code in cloudcoil/caching/_informer.py
611
612
613
def list_index_keys(self, index_name: str) -> List[str]:
    """List all keys for a given index."""
    return self._store.list_index_keys(index_name)

on_add(handler=None)

on_add(handler: EventHandler) -> None
on_add(
    handler: None = None,
) -> Callable[[EventHandler], EventHandler]

Register a handler for add events. Can be used as a decorator.

Usage

As a decorator

@informer.on_add def handle_add(obj): print(f"Added: {obj.metadata.name}")

As a regular method

informer.on_add(handle_add)

Source code in cloudcoil/caching/_informer.py
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
def on_add(self, handler: Optional[EventHandler] = None) -> Any:
    """Register a handler for add events. Can be used as a decorator.

    Usage:
        # As a decorator
        @informer.on_add
        def handle_add(obj):
            print(f"Added: {obj.metadata.name}")

        # As a regular method
        informer.on_add(handle_add)
    """

    def register(h: EventHandler) -> EventHandler:
        self._dispatcher.register_add_handler(h)
        return h

    if handler is None:
        # Being used as a decorator
        return register
    else:
        # Being called directly with a handler
        register(handler)
        return None

on_delete(handler=None)

on_delete(handler: EventHandler) -> None
on_delete(
    handler: None = None,
) -> Callable[[EventHandler], EventHandler]

Register a handler for delete events. Can be used as a decorator.

Usage

As a decorator

@informer.on_delete def handle_delete(obj): print(f"Deleted: {obj.metadata.name}")

As a regular method

informer.on_delete(handle_delete)

Source code in cloudcoil/caching/_informer.py
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
def on_delete(self, handler: Optional[EventHandler] = None) -> Any:
    """Register a handler for delete events. Can be used as a decorator.

    Usage:
        # As a decorator
        @informer.on_delete
        def handle_delete(obj):
            print(f"Deleted: {obj.metadata.name}")

        # As a regular method
        informer.on_delete(handle_delete)
    """

    def register(h: EventHandler) -> EventHandler:
        self._dispatcher.register_delete_handler(h)
        return h

    if handler is None:
        # Being used as a decorator
        return register
    else:
        # Being called directly with a handler
        register(handler)
        return None

on_update(handler=None)

on_update(handler: UpdateHandler) -> None
on_update(
    handler: None = None,
) -> Callable[[UpdateHandler], UpdateHandler]

Register a handler for update events. Can be used as a decorator.

Usage

As a decorator

@informer.on_update def handle_update(old_obj, new_obj): print(f"Updated: {new_obj.metadata.name}")

As a regular method

informer.on_update(handle_update)

Source code in cloudcoil/caching/_informer.py
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
def on_update(self, handler: Optional[UpdateHandler] = None) -> Any:
    """Register a handler for update events. Can be used as a decorator.

    Usage:
        # As a decorator
        @informer.on_update
        def handle_update(old_obj, new_obj):
            print(f"Updated: {new_obj.metadata.name}")

        # As a regular method
        informer.on_update(handle_update)
    """

    def register(h: UpdateHandler) -> UpdateHandler:
        self._dispatcher.register_update_handler(h)
        return h

    if handler is None:
        # Being used as a decorator
        return register
    else:
        # Being called directly with a handler
        register(handler)
        return None

Run controllers together, sharing compatible watches and stopping on failure.

Configs may be specified per controller, on the manager, or in the active context, in that order. Watch sharing never crosses Config instances. Each manager runs once and owns the shared informers until all workers have stopped.

Source code in cloudcoil/controller/_manager.py
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
class Manager:
    """Run controllers together, sharing compatible watches and stopping on failure.

    Configs may be specified per controller, on the manager, or in the active
    context, in that order. Watch sharing never crosses Config instances. Each
    manager runs once and owns the shared informers until all workers have stopped.
    """

    def __init__(
        self,
        *controllers: Controller[Any],
        config: Config | None = None,
        leader_election: LeaderElection | None = None,
        health: HealthServer | None = None,
        leader_lifespan: Callable[[], AbstractAsyncContextManager[None]] | None = None,
    ) -> None:
        if not controllers:
            raise ValueError("A manager needs at least one controller")
        if len({id(controller) for controller in controllers}) != len(controllers):
            raise ValueError("A controller cannot be registered twice")
        self._names = tuple(
            controller.name or f"{controller.resource.gvk().kind.lower()}-{index}"
            for index, controller in enumerate(controllers, 1)
        )
        if len(set(self._names)) != len(self._names):
            raise ValueError("Managed controllers must have distinct metric names")
        self.health = health
        self._leader_lifespan = leader_lifespan
        self._running = False
        self._controllers = controllers
        self._config = config
        self.leader_election = leader_election
        self._pool = _InformerPool()
        self._used = False
        self._finished = asyncio.Event()
        self._failure: BaseException | None = None

    @property
    def ready(self) -> bool:
        return (
            self.healthy
            and (self.leader_election is None or self.leader_election.is_leader)
            and not self._finished.is_set()
            and all(controller.ready for controller in self._controllers)
        )

    @property
    def healthy(self) -> bool:
        """Running without fatal failure, including startup and standby."""
        return self._running and self._failure is None and not self._finished.is_set()

    def metrics(self) -> str:
        """Prometheus text exposition; local counters persist after shutdown."""
        return _render(self)

    @property
    def informer_count(self) -> int:
        """Number of distinct manager-owned watch subscriptions."""
        return self._pool.count

    async def wait_ready(self, timeout: float = 30) -> None:
        async def wait_controllers() -> None:
            async with asyncio.TaskGroup() as group:
                for controller in self._controllers:
                    group.create_task(controller.wait_ready(timeout))

        ready = asyncio.create_task(wait_controllers())
        finished = asyncio.create_task(self._finished.wait())
        try:
            async with asyncio.timeout(timeout):
                await asyncio.wait((ready, finished), return_when=asyncio.FIRST_COMPLETED)
            if self._failure is not None:
                raise self._failure
            if self._finished.is_set():
                raise RuntimeError("Manager stopped before becoming ready")
            await ready
            if not self.ready:
                raise RuntimeError("Manager stopped or lost leadership before becoming ready")
        finally:
            ready.cancel()
            finished.cancel()
            await asyncio.gather(ready, finished, return_exceptions=True)

    async def _prepare(self) -> None:
        if any(
            controller._used or controller._pool is not None for controller in self._controllers
        ):
            raise RuntimeError("Controllers must be unused and belong to only one manager")
        # Register every handler before starting any shared list/watch. Otherwise a
        # late subscriber could miss initial objects and events during registration.
        for controller in self._controllers:
            controller._pool = self._pool
            config = controller.config or self._config or context.active_config
            await controller._install(config)
            controller._prepared_config = config

    async def _run_controllers(self, stop: asyncio.Event | None) -> None:
        async with AsyncExitStack() as stack:
            if self._leader_lifespan is not None:
                await stack.enter_async_context(self._leader_lifespan())
            await self._prepare()
            async with asyncio.TaskGroup() as group:
                for controller in self._controllers:
                    group.create_task(controller.run(stop=stop))

    async def run(self, *, stop: asyncio.Event | None = None) -> None:
        """Run until explicit stop, cancellation, or a fatal controller failure."""
        if self._used:
            raise RuntimeError("Manager instances can only run once")
        self._used = True
        self._running = True
        try:
            if self.health is not None:
                await self.health._start(self)
            stop = stop if stop is not None else asyncio.Event()
            if self.leader_election is None:
                if not stop.is_set():
                    await self._run_controllers(stop)
            else:
                config = (
                    self.leader_election.config
                    or self._config
                    or self._controllers[0].config
                    or context.active_config
                )
                await self.leader_election._run(
                    lambda: self._run_controllers(stop),
                    config=config,
                    stop=stop,
                )
        except BaseException as exc:
            self._failure = exc
            raise
        finally:
            self._running = False
            try:
                await self._pool.stop()
            finally:
                try:
                    if self.health is not None:
                        await self.health._close()
                finally:
                    self._finished.set()

healthy property

Running without fatal failure, including startup and standby.

informer_count property

Number of distinct manager-owned watch subscriptions.

metrics()

Prometheus text exposition; local counters persist after shutdown.

Source code in cloudcoil/controller/_manager.py
69
70
71
def metrics(self) -> str:
    """Prometheus text exposition; local counters persist after shutdown."""
    return _render(self)

run(*, stop=None) async

Run until explicit stop, cancellation, or a fatal controller failure.

Source code in cloudcoil/controller/_manager.py
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
async def run(self, *, stop: asyncio.Event | None = None) -> None:
    """Run until explicit stop, cancellation, or a fatal controller failure."""
    if self._used:
        raise RuntimeError("Manager instances can only run once")
    self._used = True
    self._running = True
    try:
        if self.health is not None:
            await self.health._start(self)
        stop = stop if stop is not None else asyncio.Event()
        if self.leader_election is None:
            if not stop.is_set():
                await self._run_controllers(stop)
        else:
            config = (
                self.leader_election.config
                or self._config
                or self._controllers[0].config
                or context.active_config
            )
            await self.leader_election._run(
                lambda: self._run_controllers(stop),
                config=config,
                stop=stop,
            )
    except BaseException as exc:
        self._failure = exc
        raise
    finally:
        self._running = False
        try:
            await self._pool.stop()
        finally:
            try:
                if self.health is not None:
                    await self.health._close()
            finally:
                self._finished.set()

Elect one active manager using a coordination.k8s.io/v1 Lease.

Identity defaults to a process-unique hostname/UUID. Explicit identities must also be unique per participant. Takeover waits for the observed record to stop changing for its lease duration, measured locally; remote wall clocks are not used for expiry. Like client-go, this is coordination, not distributed fencing. Reconciliation and external operations must remain idempotent and cancellable.

Source code in cloudcoil/controller/_leader.py
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
class LeaderElection:
    """Elect one active manager using a coordination.k8s.io/v1 Lease.

    Identity defaults to a process-unique hostname/UUID. Explicit identities must
    also be unique per participant. Takeover waits for the observed record to stop
    changing for its lease duration, measured locally; remote wall clocks are not
    used for expiry. Like client-go, this is coordination, not distributed fencing.
    Reconciliation and external operations must remain idempotent and cancellable.
    """

    def __init__(
        self,
        name: str,
        *,
        namespace: str | None = None,
        identity: str | None = None,
        lease_duration: int = 15,
        renew_deadline: float = 10,
        retry_period: float = 2,
        config: Config | None = None,
    ) -> None:
        self.name = _dns_name(name)
        self.namespace = _dns_name(namespace, namespace=True) if namespace is not None else None
        self.identity = identity if identity is not None else f"{socket.gethostname()}_{uuid4()}"
        if not self.identity or len(self.identity) > 128:
            raise ValueError("Lease identity must have 1-128 characters")
        if (
            isinstance(lease_duration, bool)
            or not isinstance(lease_duration, int)
            or not 1 <= lease_duration <= 2**31 - 1
        ):
            raise ValueError("lease_duration must be a positive int32 number of seconds")
        if (
            not all(math.isfinite(value) for value in (renew_deadline, retry_period))
            or not 0 < retry_period < renew_deadline < lease_duration
        ):
            raise ValueError("Require 0 < retry_period < renew_deadline < lease_duration")
        self.lease_duration = lease_duration
        self.renew_deadline = renew_deadline
        self.retry_period = retry_period
        self.config = config
        self._clock = time.monotonic
        self._active = False
        self._failure: BaseException | None = None
        self._used = False
        self._last_renewed = 0.0
        self._observed_at = 0.0
        self._record: str | None = None
        self._lease: dict[str, Any] | None = None
        self._acquisitions = 0
        self._renewal_failures = 0

    @property
    def is_leader(self) -> bool:
        return self._active and self._clock() < self._last_renewed + self.renew_deadline

    def _urls(self, config: Config) -> tuple[str, str]:
        namespace = _dns_name(self.namespace or config.namespace, namespace=True)
        collection = f"/apis/coordination.k8s.io/v1/namespaces/{namespace}/leases"
        return collection, f"{collection}/{self.name}"

    def _observe(self, lease: dict[str, Any]) -> None:
        metadata = lease.get("metadata") or {}
        if not metadata.get("uid") or not metadata.get("resourceVersion"):
            raise ValueError("Lease response is missing UID or resourceVersion")
        spec = lease.get("spec") or {}
        record = json.dumps(
            {
                "uid": metadata["uid"],
                **{
                    key: spec.get(key)
                    for key in (
                        "holderIdentity",
                        "leaseDurationSeconds",
                        "acquireTime",
                        "renewTime",
                        "leaseTransitions",
                    )
                },
            },
            sort_keys=True,
        )
        if record != self._record:
            self._record = record
            self._observed_at = self._clock()
        self._lease = lease

    async def _attempt(self, config: Config) -> bool:
        """Acquire/renew with one bounded GET + conditional create/update attempt."""
        started = self._clock()
        timeout = self.retry_period
        if self._active:
            timeout = min(timeout, self._last_renewed + self.renew_deadline - started)
            if timeout <= 0:
                raise LeadershipLost("Lease renewal deadline exceeded")
        try:
            async with asyncio.timeout(timeout):
                collection, url = self._urls(config)
                response = await config.async_client.get(url)
                now = datetime.now(timezone.utc).isoformat(timespec="microseconds")
                if response.status_code == 404:
                    if self._active:
                        raise LeadershipLost("Leader Lease was deleted")
                    body: dict[str, Any] = {
                        "apiVersion": "coordination.k8s.io/v1",
                        "kind": "Lease",
                        "metadata": {
                            "name": self.name,
                            "namespace": self.namespace or config.namespace,
                        },
                        "spec": {
                            "holderIdentity": self.identity,
                            "leaseDurationSeconds": self.lease_duration,
                            "acquireTime": now,
                            "renewTime": now,
                            "leaseTransitions": 0,
                        },
                    }
                    response = await config.async_client.post(collection, json=body)
                else:
                    raise_for_status(response)
                    current = response.json()
                    previous_uid = self._lease["metadata"]["uid"] if self._lease else None
                    spec = current.get("spec") or {}
                    holder = spec.get("holderIdentity")
                    if self._active and (
                        holder != self.identity or current["metadata"]["uid"] != previous_uid
                    ):
                        raise LeadershipLost("Leader Lease ownership changed")
                    self._observe(current)
                    if holder and holder != self.identity:
                        duration = spec.get("leaseDurationSeconds")
                        if (
                            isinstance(duration, bool)
                            or not isinstance(duration, int)
                            or duration <= 0
                        ):
                            raise ValueError("Held Lease has an invalid leaseDurationSeconds")
                        if self._clock() - self._observed_at < duration:
                            return False
                    body = deepcopy(current)
                    body["spec"] = {
                        **spec,
                        "holderIdentity": self.identity,
                        "leaseDurationSeconds": self.lease_duration,
                        "renewTime": now,
                    }
                    if holder != self.identity:
                        body["spec"]["acquireTime"] = now
                        body["spec"]["leaseTransitions"] = (spec.get("leaseTransitions") or 0) + 1
                    # resourceVersion from the GET prevents two contenders winning.
                    response = await config.async_client.put(url, json=body)
                raise_for_status(response)
                self._observe(response.json())
                # Use request start, not response arrival, as the conservative local deadline.
                self._last_renewed = started
                return True
        except APIError as exc:
            if (
                exc.status_code is not None
                and 400 <= exc.status_code < 500
                and exc.status_code not in {404, 408, 409, 429}
            ):
                raise
            logger.warning("Lease %s request failed: %s", self.name, exc)
        except (httpx.RequestError, TimeoutError) as exc:
            logger.warning("Lease %s request failed: %s", self.name, exc)
        return False

    async def _renew(self, config: Config) -> None:
        while True:
            remaining = self._last_renewed + self.renew_deadline - self._clock()
            if remaining <= 0:
                raise LeadershipLost("Lease renewal deadline exceeded")
            await asyncio.sleep(min(self.retry_period, remaining))
            if not await self._attempt(config):
                self._renewal_failures += 1

    async def _release(self, config: Config) -> None:
        if self._lease is None:
            return
        try:
            async with asyncio.timeout(self.retry_period):
                _, url = self._urls(config)
                response = await config.async_client.get(url)
                if response.status_code == 404:
                    return
                raise_for_status(response)
                current = response.json()
                if (current.get("spec") or {}).get("holderIdentity") != self.identity or current[
                    "metadata"
                ]["uid"] != self._lease["metadata"]["uid"]:
                    return
                body = deepcopy(current)
                body["spec"]["holderIdentity"] = ""
                body["spec"]["leaseDurationSeconds"] = 1
                body["spec"]["renewTime"] = datetime.now(timezone.utc).isoformat(
                    timespec="microseconds"
                )
                # Never clear a successor's ownership, even if it changes after this GET.
                response = await config.async_client.put(url, json=body)
                raise_for_status(response)
        except (APIError, httpx.RequestError, TimeoutError) as exc:
            logger.warning("Could not release Lease %s; it will expire: %s", self.name, exc)

    async def _run(
        self,
        callback: Callable[[], Awaitable[None]],
        *,
        config: Config,
        stop: asyncio.Event,
    ) -> None:
        if self._used:
            raise RuntimeError("LeaderElection instances can only run once")
        self._used = True
        work: asyncio.Task[None] | None = None
        renew: asyncio.Task[None] | None = None
        acquired = False
        try:
            while not stop.is_set():
                if await self._attempt(config):
                    acquired = True
                    break
                try:
                    await asyncio.wait_for(
                        stop.wait(), self.retry_period * random.uniform(0.9, 1.1)
                    )
                except TimeoutError:
                    pass
            if not acquired or stop.is_set():
                return
            self._active = True
            self._acquisitions += 1
            logger.info("Acquired Lease %s as %s", self.name, self.identity)

            async def invoke() -> None:
                await callback()

            work = asyncio.create_task(invoke())
            renew = asyncio.create_task(self._renew(config))
            done, _ = await asyncio.wait((work, renew), return_when=asyncio.FIRST_COMPLETED)
            if renew in done:
                await renew
                raise LeadershipLost("Lease renewal stopped unexpectedly")
            await work
        except BaseException as error:
            self._failure = error
            raise
        finally:
            self._active = False
            # Cancelling/joining the callback also joins its controller workers. Release
            # only afterwards, so a healthy successor cannot race draining workers.
            tasks = [task for task in (work, renew) if task is not None]
            for task in tasks:
                task.cancel()
            outcomes = await asyncio.gather(*tasks, return_exceptions=True)
            if acquired:
                await self._release(config)
            cleanup_errors = [
                outcome
                for outcome in outcomes
                if isinstance(outcome, BaseException)
                and not isinstance(outcome, asyncio.CancelledError)
                and outcome is not self._failure
            ]
            if cleanup_errors:
                if self._failure is not None and not isinstance(
                    self._failure, asyncio.CancelledError
                ):
                    cleanup_errors.insert(0, self._failure)
                raise BaseExceptionGroup("Leader work failed during cleanup", cleanup_errors)

Serve GET /healthz, /readyz, and /metrics; opt in through Manager(health=...).

Defaults to loopback. Bind host="0.0.0.0" for container probes. This is a plain HTTP diagnostics listener; protect access with your network configuration. Connections have bounded headers, a read deadline, and no keep-alive.

Source code in cloudcoil/controller/_health.py
 11
 12
 13
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
class HealthServer:
    """Serve GET /healthz, /readyz, and /metrics; opt in through Manager(health=...).

    Defaults to loopback. Bind host="0.0.0.0" for container probes. This is a plain
    HTTP diagnostics listener; protect access with your network configuration.
    Connections have bounded headers, a read deadline, and no keep-alive.
    """

    def __init__(self, *, host: str = "127.0.0.1", port: int = 8080) -> None:
        if isinstance(port, bool) or not isinstance(port, int) or not 0 <= port <= 65535:
            raise ValueError("port must be an integer between 0 and 65535")
        self.host = host
        self.port = port
        self._server: asyncio.Server | None = None
        self._tasks: set[asyncio.Task[None]] = set()
        self._used = False
        self._closing = False

    @property
    def address(self) -> tuple[str, int] | None:
        """Bound host and port (including an assigned port=0), or None when stopped."""
        if self._server is None or not self._server.sockets:
            return None
        host, port, *_ = self._server.sockets[0].getsockname()
        return host, port

    async def _start(self, manager: "Manager") -> None:
        if self._used:
            raise RuntimeError("HealthServer instances can only run once")
        self._used = True

        def connected(reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
            if self._closing or len(self._tasks) >= 128:
                writer.close()
                return
            task = asyncio.create_task(self._handle(manager, reader, writer))
            self._tasks.add(task)

            def completed(task: asyncio.Task[None]) -> None:
                # A connection can be cancelled before _handle enters its finally.
                writer.close()
                self._tasks.discard(task)

            task.add_done_callback(completed)

        self._server = await asyncio.start_server(connected, self.host, self.port, limit=8192)

    async def _handle(
        self, manager: "Manager", reader: asyncio.StreamReader, writer: asyncio.StreamWriter
    ) -> None:
        try:
            code, body = HTTPStatus.BAD_REQUEST, "bad request\n"
            content_type = "text/plain; charset=utf-8"
            try:
                async with asyncio.timeout(5):
                    headers = await reader.readuntil(b"\r\n\r\n")
                first = headers.split(b"\r\n", 1)[0].split(b" ")
                if len(first) == 3 and first[2] in (b"HTTP/1.0", b"HTTP/1.1"):
                    method, path, _ = first
                    if method != b"GET":
                        code, body = HTTPStatus.METHOD_NOT_ALLOWED, "method not allowed\n"
                    elif path == b"/healthz":
                        code = HTTPStatus.OK if manager.healthy else HTTPStatus.SERVICE_UNAVAILABLE
                        body = "ok\n" if manager.healthy else "unhealthy\n"
                    elif path == b"/readyz":
                        code = HTTPStatus.OK if manager.ready else HTTPStatus.SERVICE_UNAVAILABLE
                        body = "ok\n" if manager.ready else "not ready\n"
                    elif path == b"/metrics":
                        code, body = HTTPStatus.OK, manager.metrics()
                        content_type = "text/plain; version=0.0.4; charset=utf-8"
                    else:
                        code, body = HTTPStatus.NOT_FOUND, "not found\n"
            except asyncio.LimitOverrunError:
                code, body = HTTPStatus.REQUEST_HEADER_FIELDS_TOO_LARGE, "headers too large\n"
            except TimeoutError:
                code, body = HTTPStatus.REQUEST_TIMEOUT, "request timeout\n"
            except asyncio.IncompleteReadError:
                pass
            payload = body.encode("utf-8")
            response = (
                f"HTTP/1.1 {code.value} {code.phrase}\r\n"
                f"Content-Type: {content_type}\r\nContent-Length: {len(payload)}\r\n"
                "Connection: close\r\nCache-Control: no-store\r\n"
                + ("Allow: GET\r\n" if code == HTTPStatus.METHOD_NOT_ALLOWED else "")
                + "\r\n"
            ).encode("ascii") + payload
            writer.write(response)
            async with asyncio.timeout(5):
                await writer.drain()
        except ConnectionError, TimeoutError:
            pass
        finally:
            writer.close()
            try:
                async with asyncio.timeout(1):
                    await writer.wait_closed()
            except ConnectionError, TimeoutError:
                pass

    async def _close(self) -> None:
        self._closing = True
        if self._server is not None:
            self._server.close()
            # Cancel clients before wait_closed, which may wait for active streams.
            tasks = list(self._tasks)
            for task in tasks:
                task.cancel()
            await asyncio.gather(*tasks, return_exceptions=True)
            await self._server.wait_closed()
            self._server = None

address property

Bound host and port (including an assigned port=0), or None when stopped.

Immutable local snapshot; counts reset when a controller is recreated.

Source code in cloudcoil/controller/_metrics.py
13
14
15
16
17
18
19
20
21
22
23
24
25
@dataclass(frozen=True)
class ControllerStatus:
    """Immutable local snapshot; counts reset when a controller is recreated."""

    ready: bool
    queued: int
    processing: int
    delayed: int
    successes: int
    errors: int
    terminal_errors: int
    cancellations: int
    duration_seconds: float

Serialize each key while allowing different keys to run concurrently.

Call add/done/retry from the same event loop as get. Repeated adds coalesce; an add during processing guarantees one more pass after done. Delayed work uses one timer per key, never a sleeping task per event. This is an in-memory queue: restart recovery comes from listing the desired Kubernetes state.

Source code in cloudcoil/controller/_queue.py
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
class WorkQueue[K: Hashable]:
    """Serialize each key while allowing different keys to run concurrently.

    Call add/done/retry from the same event loop as get. Repeated adds coalesce;
    an add during processing guarantees one more pass after done. Delayed work
    uses one timer per key, never a sleeping task per event. This is an in-memory
    queue: restart recovery comes from listing the desired Kubernetes state.
    """

    def __init__(
        self, *, base_delay: float = 1.0, max_delay: float = 60.0, jitter: float = 0.1
    ) -> None:
        if not all(math.isfinite(v) for v in (base_delay, max_delay, jitter)):
            raise ValueError("Retry settings must be finite")
        if base_delay <= 0 or max_delay < base_delay or not 0 <= jitter <= 1:
            raise ValueError("Require 0 < base_delay <= max_delay and 0 <= jitter <= 1")
        self._base_delay = base_delay
        self._max_delay = max_delay
        self._jitter = jitter
        self._ready: deque[K] = deque()
        self._dirty: set[K] = set()
        self._processing: set[K] = set()
        self._delayed: dict[K, asyncio.TimerHandle] = {}
        self._retries: dict[K, int] = {}
        self._available = asyncio.Event()
        self._idle = asyncio.Event()
        self._idle.set()
        self._closed = False
        self._immediate = False

    @property
    def depth(self) -> int:
        """Ready keys waiting for a worker; excludes in-flight and delayed keys."""
        return len(self._ready)

    @property
    def processing(self) -> int:
        """Keys currently reserved by workers."""
        return len(self._processing)

    @property
    def delayed(self) -> int:
        """Keys with a pending retry or scheduled requeue."""
        return len(self._delayed)

    def add(self, key: K) -> None:
        """Request reconciliation now; fresh events supersede a delayed retry."""
        if self._closed:
            return
        timer = self._delayed.pop(key, None)
        if timer is not None:
            timer.cancel()
        if key in self._dirty:
            return
        self._idle.clear()
        self._dirty.add(key)
        if key not in self._processing:
            self._ready.append(key)
            self._available.set()

    def add_after(self, key: K, delay: float) -> None:
        """Schedule a key, preserving the earliest pending deadline."""
        if not math.isfinite(delay) or delay < 0:
            raise ValueError("delay must be finite and nonnegative")
        if self._closed:
            return
        if delay == 0:
            self.add(key)
            return
        if key in self._dirty:
            return
        loop = asyncio.get_running_loop()
        deadline = loop.time() + delay
        current = self._delayed.get(key)
        if current is not None:
            if current.when() <= deadline:
                return
            current.cancel()
        self._idle.clear()
        self._delayed[key] = loop.call_at(deadline, self._release, key)

    def _release(self, key: K) -> None:
        self._delayed.pop(key, None)
        self.add(key)

    def retry(self, key: K) -> float:
        """Schedule exponential backoff with jitter and return the chosen delay."""
        if self._closed:
            return 0.0
        attempt = self._retries.get(key, 0)
        self._retries[key] = attempt + 1
        # Clamp the exponent before computing it, even after prolonged failures.
        cap = math.ceil(math.log2(self._max_delay) - math.log2(self._base_delay))
        delay = self._max_delay if attempt >= cap else math.ldexp(self._base_delay, attempt)
        delay = min(self._max_delay, delay * random.uniform(1 - self._jitter, 1 + self._jitter))
        self.add_after(key, delay)
        return delay

    def forget(self, key: K) -> None:
        """Reset failure history after success or a terminal error."""
        self._retries.pop(key, None)

    def num_retries(self, key: K) -> int:
        """Return consecutive retry requests since the last forget."""
        return self._retries.get(key, 0)

    async def get(self) -> K:
        """Take the next key; always pair a successful get with done in finally."""
        while not self._ready:
            if self._closed:
                raise QueueClosed
            self._available.clear()
            await self._available.wait()
        key = self._ready.popleft()
        self._dirty.remove(key)
        self._processing.add(key)
        return key

    def done(self, key: K) -> None:
        """Release a processing key and queue any update that arrived meanwhile."""
        if key not in self._processing:
            raise ValueError("done called for a key that is not processing")
        self._processing.remove(key)
        if key in self._dirty and not self._immediate:
            self._ready.append(key)
            self._available.set()
        self._check_idle()

    async def join(self) -> None:
        """Wait until ready, processing, and delayed work are all finished."""
        await self._idle.wait()

    def shutdown(self, *, immediate: bool = False) -> None:
        """Reject new work and discard timers; optionally discard ready work too.

        With immediate=False, consumers can drain accepted ready/dirty work.
        In-flight work always requires done; shutdown never cancels callers.
        """
        self._closed = True
        self._immediate = self._immediate or immediate
        for timer in self._delayed.values():
            timer.cancel()
        self._delayed.clear()
        self._retries.clear()
        if self._immediate:
            self._ready.clear()
            self._dirty.clear()
        self._available.set()
        self._check_idle()

    def _check_idle(self) -> None:
        if not (self._dirty or self._processing or self._delayed):
            self._idle.set()

delayed property

Keys with a pending retry or scheduled requeue.

depth property

Ready keys waiting for a worker; excludes in-flight and delayed keys.

processing property

Keys currently reserved by workers.

add(key)

Request reconciliation now; fresh events supersede a delayed retry.

Source code in cloudcoil/controller/_queue.py
59
60
61
62
63
64
65
66
67
68
69
70
71
72
def add(self, key: K) -> None:
    """Request reconciliation now; fresh events supersede a delayed retry."""
    if self._closed:
        return
    timer = self._delayed.pop(key, None)
    if timer is not None:
        timer.cancel()
    if key in self._dirty:
        return
    self._idle.clear()
    self._dirty.add(key)
    if key not in self._processing:
        self._ready.append(key)
        self._available.set()

add_after(key, delay)

Schedule a key, preserving the earliest pending deadline.

Source code in cloudcoil/controller/_queue.py
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
def add_after(self, key: K, delay: float) -> None:
    """Schedule a key, preserving the earliest pending deadline."""
    if not math.isfinite(delay) or delay < 0:
        raise ValueError("delay must be finite and nonnegative")
    if self._closed:
        return
    if delay == 0:
        self.add(key)
        return
    if key in self._dirty:
        return
    loop = asyncio.get_running_loop()
    deadline = loop.time() + delay
    current = self._delayed.get(key)
    if current is not None:
        if current.when() <= deadline:
            return
        current.cancel()
    self._idle.clear()
    self._delayed[key] = loop.call_at(deadline, self._release, key)

done(key)

Release a processing key and queue any update that arrived meanwhile.

Source code in cloudcoil/controller/_queue.py
132
133
134
135
136
137
138
139
140
def done(self, key: K) -> None:
    """Release a processing key and queue any update that arrived meanwhile."""
    if key not in self._processing:
        raise ValueError("done called for a key that is not processing")
    self._processing.remove(key)
    if key in self._dirty and not self._immediate:
        self._ready.append(key)
        self._available.set()
    self._check_idle()

forget(key)

Reset failure history after success or a terminal error.

Source code in cloudcoil/controller/_queue.py
112
113
114
def forget(self, key: K) -> None:
    """Reset failure history after success or a terminal error."""
    self._retries.pop(key, None)

get() async

Take the next key; always pair a successful get with done in finally.

Source code in cloudcoil/controller/_queue.py
120
121
122
123
124
125
126
127
128
129
130
async def get(self) -> K:
    """Take the next key; always pair a successful get with done in finally."""
    while not self._ready:
        if self._closed:
            raise QueueClosed
        self._available.clear()
        await self._available.wait()
    key = self._ready.popleft()
    self._dirty.remove(key)
    self._processing.add(key)
    return key

join() async

Wait until ready, processing, and delayed work are all finished.

Source code in cloudcoil/controller/_queue.py
142
143
144
async def join(self) -> None:
    """Wait until ready, processing, and delayed work are all finished."""
    await self._idle.wait()

num_retries(key)

Return consecutive retry requests since the last forget.

Source code in cloudcoil/controller/_queue.py
116
117
118
def num_retries(self, key: K) -> int:
    """Return consecutive retry requests since the last forget."""
    return self._retries.get(key, 0)

retry(key)

Schedule exponential backoff with jitter and return the chosen delay.

Source code in cloudcoil/controller/_queue.py
 99
100
101
102
103
104
105
106
107
108
109
110
def retry(self, key: K) -> float:
    """Schedule exponential backoff with jitter and return the chosen delay."""
    if self._closed:
        return 0.0
    attempt = self._retries.get(key, 0)
    self._retries[key] = attempt + 1
    # Clamp the exponent before computing it, even after prolonged failures.
    cap = math.ceil(math.log2(self._max_delay) - math.log2(self._base_delay))
    delay = self._max_delay if attempt >= cap else math.ldexp(self._base_delay, attempt)
    delay = min(self._max_delay, delay * random.uniform(1 - self._jitter, 1 + self._jitter))
    self.add_after(key, delay)
    return delay

shutdown(*, immediate=False)

Reject new work and discard timers; optionally discard ready work too.

With immediate=False, consumers can drain accepted ready/dirty work. In-flight work always requires done; shutdown never cancels callers.

Source code in cloudcoil/controller/_queue.py
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
def shutdown(self, *, immediate: bool = False) -> None:
    """Reject new work and discard timers; optionally discard ready work too.

    With immediate=False, consumers can drain accepted ready/dirty work.
    In-flight work always requires done; shutdown never cancels callers.
    """
    self._closed = True
    self._immediate = self._immediate or immediate
    for timer in self._delayed.values():
        timer.cancel()
    self._delayed.clear()
    self._retries.clear()
    if self._immediate:
        self._ready.clear()
        self._dirty.clear()
    self._available.set()
    self._check_idle()

Explicit reconciliation and writes

These APIs support embedding and existing Request callbacks. Start with the decorator guide for new applications.

Typed asynchronous Kubernetes reconciliation and controller lifecycle.

Request dataclass

Latest cached state at worker dispatch, copied so mutation cannot corrupt the cache.

resource=None means the key is absent from the watched scope (deleted or no longer selected). It is not a deletion proof for destructive external cleanup; use a finalizer and a live API read for that. Reads are eventually consistent.

Source code in cloudcoil/controller/_types.py
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
@dataclass(frozen=True)
class Request[T: Resource]:
    """Latest cached state at worker dispatch, copied so mutation cannot corrupt the cache.

    resource=None means the key is absent from the watched scope (deleted or no
    longer selected). It is not a deletion proof for destructive external cleanup;
    use a finalizer and a live API read for that. Reads are eventually consistent.
    """

    key: ResourceKey
    resource: T | None
    config: "Config | None" = None
    _informers: "dict[type[Resource], AsyncInformer[Any]]" = field(
        default_factory=dict, repr=False, compare=False
    )
    _events: "EventRecorder | None" = field(default=None, repr=False, compare=False)
    _report: _Report = field(default_factory=_Report, repr=False, compare=False)

    @property
    def object(self) -> T:
        """The present primary object; stages are only invoked for present objects."""
        current = self._report.current if self._report.current is not None else self.resource
        if current is None:
            raise ValueError("The primary resource is absent from the watched scope")
        return cast(T, current)

    def set_status(self, **changes: Any) -> None:
        """Stage validated status fields for persistence, even if the handler fails.

        Only explicit helper updates are saved on failure, never spec or metadata.
        Ordinary resource edits still require returning the resource on success.
        """
        update_status(self.object, **changes)
        self._report.status = self.object.status.model_copy(deep=True)  # type: ignore[attr-defined]
        self._report.dirty = True

    def condition(
        self,
        condition: str,
        status: bool | Literal["True", "False", "Unknown"],
        *,
        reason: str,
        message: str = "",
        event: bool = False,
        warning: bool = False,
        action: str | None = None,
    ) -> None:
        """Stage a standard condition; optionally emit an Event on a transition.

        A changed truth value or reason counts as an Event transition. Message and
        generation-only changes do not. Events flush only after status persistence.
        """
        previous = get_condition(self.object, condition)
        set_condition(self.object, condition, status, reason=reason, message=message)
        self.set_status()
        current = get_condition(self.object, condition)
        assert current is not None
        if event and (
            previous is None
            or (previous.status, previous.reason) != (current.status, current.reason)
        ):
            self._report.events.append(
                (reason, message, "Warning" if warning else "Normal", action or condition)
            )

    def _failed(self, error: Exception) -> None:
        """Report stable, non-sensitive failure details for the active stage."""
        report = self._report
        name = report.pending[0] if report.pending else "Reconcile"
        reason = "TerminalError" if isinstance(error, TerminalError) else "ReconcileFailed"
        message = f"{name} failed ({type(error).__name__}); see controller logs"
        if report.managed:
            if report.pending and report.stage_conditions:
                self.condition(
                    name, False, reason=reason, message=message, event=True, warning=True
                )
            for pending in report.pending[1:] if report.stage_conditions else []:
                self.condition(pending, "Unknown", reason="DependencyNotReady")
            self.condition(
                "Ready",
                False,
                reason=reason,
                message=message,
                event=not report.stage_conditions,
                warning=True,
                action=name,
            )
        elif report.pending:
            report.events.append((reason, message, "Warning", name))

    async def event(
        self, reason: str, message: str, *, type: Literal["Normal", "Warning"] = "Normal"
    ) -> bool:
        """Record a bounded, best-effort Kubernetes Event regarding this resource."""
        if self.resource is None or self._events is None:
            return False
        return await self._events.emit(
            self.resource, reason, message, type=type, config=self.config
        )

    def cached[U: Resource](self, resource: type[U]) -> CachedResources[U]:
        """Read the primary or a declared .owns/.watch informer, without I/O."""
        if resource not in self._informers:
            raise ValueError(f"{resource.__name__} is not watched by this controller")
        return CachedResources(cast("AsyncInformer[U]", self._informers[resource]), self.namespace)

    async def client[U: Resource](self, resource: type[U]) -> "AsyncAPIClient[U]":
        """A live client for any kind, sharing this operator's connection.

        Namespaced clients default to this request's namespace. Pass a namespace
        to client operations for cross-namespace reads. Clients share the Config
        lifetime and must not be closed by handlers.
        """
        return await resource.async_client(self.config, namespace=self.namespace, cached=False)

    async def ensure[U: Resource](self, desired: U) -> U:
        """Create or patch an owned child; omitted fields remain untouched.

        Defaults name and namespace from the parent. Refuses unrelated existing
        objects. Maps merge, lists replace, and explicit None removes a field.
        Child events are subscribed separately with Controller.owns(...).
        """
        from ._children import ensure

        if self.resource is None:
            raise ValueError("Cannot ensure a child for an absent parent")
        return await ensure(self.object, desired, config=self.config)

    @property
    def name(self) -> str:
        return self.key.name

    @property
    def namespace(self) -> str | None:
        return self.key.namespace

object property

The present primary object; stages are only invoked for present objects.

cached(resource)

Read the primary or a declared .owns/.watch informer, without I/O.

Source code in cloudcoil/controller/_types.py
151
152
153
154
155
def cached[U: Resource](self, resource: type[U]) -> CachedResources[U]:
    """Read the primary or a declared .owns/.watch informer, without I/O."""
    if resource not in self._informers:
        raise ValueError(f"{resource.__name__} is not watched by this controller")
    return CachedResources(cast("AsyncInformer[U]", self._informers[resource]), self.namespace)

client(resource) async

A live client for any kind, sharing this operator's connection.

Namespaced clients default to this request's namespace. Pass a namespace to client operations for cross-namespace reads. Clients share the Config lifetime and must not be closed by handlers.

Source code in cloudcoil/controller/_types.py
157
158
159
160
161
162
163
164
async def client[U: Resource](self, resource: type[U]) -> "AsyncAPIClient[U]":
    """A live client for any kind, sharing this operator's connection.

    Namespaced clients default to this request's namespace. Pass a namespace
    to client operations for cross-namespace reads. Clients share the Config
    lifetime and must not be closed by handlers.
    """
    return await resource.async_client(self.config, namespace=self.namespace, cached=False)

condition(condition, status, *, reason, message='', event=False, warning=False, action=None)

Stage a standard condition; optionally emit an Event on a transition.

A changed truth value or reason counts as an Event transition. Message and generation-only changes do not. Events flush only after status persistence.

Source code in cloudcoil/controller/_types.py
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
def condition(
    self,
    condition: str,
    status: bool | Literal["True", "False", "Unknown"],
    *,
    reason: str,
    message: str = "",
    event: bool = False,
    warning: bool = False,
    action: str | None = None,
) -> None:
    """Stage a standard condition; optionally emit an Event on a transition.

    A changed truth value or reason counts as an Event transition. Message and
    generation-only changes do not. Events flush only after status persistence.
    """
    previous = get_condition(self.object, condition)
    set_condition(self.object, condition, status, reason=reason, message=message)
    self.set_status()
    current = get_condition(self.object, condition)
    assert current is not None
    if event and (
        previous is None
        or (previous.status, previous.reason) != (current.status, current.reason)
    ):
        self._report.events.append(
            (reason, message, "Warning" if warning else "Normal", action or condition)
        )

ensure(desired) async

Create or patch an owned child; omitted fields remain untouched.

Defaults name and namespace from the parent. Refuses unrelated existing objects. Maps merge, lists replace, and explicit None removes a field. Child events are subscribed separately with Controller.owns(...).

Source code in cloudcoil/controller/_types.py
166
167
168
169
170
171
172
173
174
175
176
177
async def ensure[U: Resource](self, desired: U) -> U:
    """Create or patch an owned child; omitted fields remain untouched.

    Defaults name and namespace from the parent. Refuses unrelated existing
    objects. Maps merge, lists replace, and explicit None removes a field.
    Child events are subscribed separately with Controller.owns(...).
    """
    from ._children import ensure

    if self.resource is None:
        raise ValueError("Cannot ensure a child for an absent parent")
    return await ensure(self.object, desired, config=self.config)

event(reason, message, *, type='Normal') async

Record a bounded, best-effort Kubernetes Event regarding this resource.

Source code in cloudcoil/controller/_types.py
141
142
143
144
145
146
147
148
149
async def event(
    self, reason: str, message: str, *, type: Literal["Normal", "Warning"] = "Normal"
) -> bool:
    """Record a bounded, best-effort Kubernetes Event regarding this resource."""
    if self.resource is None or self._events is None:
        return False
    return await self._events.emit(
        self.resource, reason, message, type=type, config=self.config
    )

set_status(**changes)

Stage validated status fields for persistence, even if the handler fails.

Only explicit helper updates are saved on failure, never spec or metadata. Ordinary resource edits still require returning the resource on success.

Source code in cloudcoil/controller/_types.py
77
78
79
80
81
82
83
84
85
def set_status(self, **changes: Any) -> None:
    """Stage validated status fields for persistence, even if the handler fails.

    Only explicit helper updates are saved on failure, never spec or metadata.
    Ordinary resource edits still require returning the resource on success.
    """
    update_status(self.object, **changes)
    self._report.status = self.object.status.model_copy(deep=True)  # type: ignore[attr-defined]
    self._report.dirty = True

Stage dataclass

An idempotent action named by the condition it establishes.

Source code in cloudcoil/controller/_stages.py
17
18
19
20
21
22
23
24
25
26
27
28
29
30
@dataclass(frozen=True)
class Stage[T: Resource]:
    """An idempotent action named by the condition it establishes."""

    name: str
    run: Step[T]

    def __post_init__(self) -> None:
        if not re.fullmatch(r"[A-Za-z][A-Za-z0-9_]*", self.name) or len(self.name) > 128:
            raise ValueError("Stage name must be an identifier of at most 128 characters")
        if self.name == "Ready":
            raise ValueError("Ready is reserved for the aggregate condition")
        if not callable(self.run):
            raise TypeError("Stage run must be an async callable")

Stages

Run stages in explicit order, stopping on waiting or failure.

Every pass starts from the first stage, including when previously Ready. Conditions are observations, never checkpoints that skip drift repair. The controller persists status on waits/errors and applies its normal retries.

Source code in cloudcoil/controller/_stages.py
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
class Stages[T: Resource]:
    """Run stages in explicit order, stopping on waiting or failure.

    Every pass starts from the first stage, including when previously Ready.
    Conditions are observations, never checkpoints that skip drift repair.
    The controller persists status on waits/errors and applies its normal retries.
    """

    def __init__(self, *stages: Stage[T], report_status: bool = True) -> None:
        if not stages or any(not isinstance(stage, Stage) for stage in stages):
            raise ValueError("Stages requires one or more Stage values")
        if len({stage.name for stage in stages}) != len(stages):
            raise ValueError("Stage names must be unique")
        self.stages = stages
        self.report_status = report_status

    def _freeze(self, resource: type[T]) -> None:
        _validate(resource, self.report_status)

    async def __call__(self, request: Request[T]) -> Result | None:
        if not _start(request, [stage.name for stage in self.stages], self.report_status):
            return None
        for stage in self.stages:
            wait = await _run(request, stage)
            if wait is not None:
                return _done(request, wait)
        return _done(request)

Cases

Register first-match cases on an instance before the controller starts.

Without priorities, registration order is execution order. When priorities are supplied, every case must have a distinct integer priority (higher runs first). Predicates are synchronous, side-effect-free reads of current state. They are evaluated lazily; only the first matching action executes, even if it waits.

Source code in cloudcoil/controller/_stages.py
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
class Cases[T: Resource]:
    """Register first-match cases on an instance before the controller starts.

    Without priorities, registration order is execution order. When priorities are
    supplied, every case must have a distinct integer priority (higher runs first).
    Predicates are synchronous, side-effect-free reads of current state. They are
    evaluated lazily; only the first matching action executes, even if it waits.
    """

    def __init__(self, *, report_status: bool = True) -> None:
        self.report_status = report_status
        self._cases: list[_Case[T]] = []
        self._otherwise: Stage[T] | None = None
        self._frozen = False

    def case(
        self, name: str, *, when: Predicate[T], priority: int | None = None
    ) -> Callable[[Step[T]], Step[T]]:
        if priority is not None and (isinstance(priority, bool) or not isinstance(priority, int)):
            raise ValueError("Case priority must be an integer")
        if not callable(when):
            raise TypeError("Case when must be a synchronous predicate")

        def register(run: Step[T]) -> Step[T]:
            self._check_name(name)
            if priority is not None and any(case.priority == priority for case in self._cases):
                raise ValueError("Case priorities must be unique; import order must not break ties")
            self._cases.append(_Case(Stage(name, run), when, priority))
            return run

        return register

    def otherwise(self, name: str) -> Callable[[Step[T]], Step[T]]:
        """Register the explicit fallback, which always follows all predicates."""

        def register(run: Step[T]) -> Step[T]:
            self._check_name(name)
            if self._otherwise is not None:
                raise ValueError("Only one otherwise case may be registered")
            self._otherwise = Stage(name, run)
            return run

        return register

    def _check_name(self, name: str) -> None:
        if self._frozen:
            raise RuntimeError("Register cases before running the controller")
        if any(case.stage.name == name for case in self._cases) or (
            self._otherwise is not None and self._otherwise.name == name
        ):
            raise ValueError("Case names must be unique")

    def _freeze(self, resource: type[T]) -> None:
        _validate(resource, self.report_status)
        if not self._cases and self._otherwise is None:
            raise ValueError("Register at least one case")
        if any(case.priority is not None for case in self._cases):
            if any(case.priority is None for case in self._cases):
                raise ValueError("Give every case a priority, or use registration order throughout")
            self._cases.sort(key=lambda case: case.priority or 0, reverse=True)
        self._frozen = True

    async def __call__(self, request: Request[T]) -> Result | None:
        if request.resource is None:
            return None
        if not self._frozen:
            self._freeze(type(request.object))
        names = [case.stage.name for case in self._cases]
        if self._otherwise is not None:
            names.append(self._otherwise.name)
        if not _start(request, names, self.report_status, stage_conditions=False):
            return None
        selected = self._otherwise
        for case in self._cases:
            request._report.pending = [
                case.stage.name,
                *[name for name in names if name != case.stage.name],
            ]
            matches = case.when(request)
            if type(matches) is not bool:
                if inspect.iscoroutine(matches):
                    matches.close()
                raise TypeError("Case predicates must return bool and must not perform async I/O")
            if matches:
                selected = case.stage
                break
        if selected is None:
            raise TerminalError(
                "No case matched; register an otherwise handler if this is expected"
            )
        # A branch name describes an action, not a Kubernetes condition type.
        # Report its outcome through Ready and retain the name as the Event action.
        request._report.pending = [selected.name]
        return _done(request, await _run(request, selected))

otherwise(name)

Register the explicit fallback, which always follows all predicates.

Source code in cloudcoil/controller/_stages.py
168
169
170
171
172
173
174
175
176
177
178
def otherwise(self, name: str) -> Callable[[Step[T]], Step[T]]:
    """Register the explicit fallback, which always follows all predicates."""

    def register(run: Step[T]) -> Step[T]:
        self._check_name(name)
        if self._otherwise is not None:
            raise ValueError("Only one otherwise case may be registered")
        self._otherwise = Stage(name, run)
        return run

    return register

mutate(resource, change, *, status=False, config=None) async

Fetch live state, edit a copy, and patch only changes with UID/version tests.

A no-op issues no PATCH. The synchronous callback must only edit the copy and have no external side effects. Conflicts propagate to the reconcile retry loop, which will read fresh state on its next attempt. The supplied resource must have a UID; never apply work for a deleted object to a replacement with the same name.

Source code in cloudcoil/controller/_mutations.py
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
async def mutate[T: Resource](
    resource: T, change: Callable[[T], None], *, status: bool = False, config: Config | None = None
) -> T:
    """Fetch live state, edit a copy, and patch only changes with UID/version tests.

    A no-op issues no PATCH. The synchronous callback must only edit the copy and
    have no external side effects. Conflicts propagate to the reconcile retry loop,
    which will read fresh state on its next attempt. The supplied resource must have
    a UID; never apply work for a deleted object to a replacement with the same name.
    """
    if not resource.name or not resource.metadata or not resource.metadata.uid:
        raise ValueError("mutate requires a fetched resource with metadata.name and UID")
    config = config or context.active_config
    client = await config.async_client_for(type(resource), cached=False)
    current = await client.get(resource.name, resource.namespace)
    if not current.metadata or current.metadata.uid != resource.metadata.uid:
        raise ResourceConflict(
            "Resource was replaced; refusing to mutate the new UID", status_code=409
        )
    desired = current.model_copy(deep=True)
    callback: Callable[[T], object] = change
    result = callback(desired)
    if result is not None:
        if inspect.iscoroutine(result):
            result.close()
        raise TypeError("Mutation callbacks edit in place and must return None")
    operations = diff(current, desired)
    if status and any(
        op["op"] != "test" and op["path"] != "/status" and not op["path"].startswith("/status/")
        for op in operations
    ):
        raise ValueError("A status mutation can only change status fields")
    if not operations:
        return current
    return await client.patch(current, operations, subresource="status" if status else None)

ensure_finalizer(resource, finalizer, *, config=None) async

Persist this controller's finalizer before creating external resources.

Refuses to add a missing finalizer once deletion has begun. Existing finalizers are preserved. Complete cleanup before calling remove_finalizer.

Source code in cloudcoil/controller/_mutations.py
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
async def ensure_finalizer[T: Resource](
    resource: T, finalizer: str, *, config: Config | None = None
) -> T:
    """Persist this controller's finalizer before creating external resources.

    Refuses to add a missing finalizer once deletion has begun. Existing finalizers
    are preserved. Complete cleanup before calling remove_finalizer.
    """
    if not finalizer or "/" not in finalizer:
        raise ValueError("Use a qualified finalizer name such as example.com/cleanup")

    def change(obj: T) -> None:
        assert obj.metadata is not None
        existing = obj.metadata.finalizers or []
        if finalizer in existing:
            return
        if obj.metadata.deletion_timestamp is not None:
            raise TerminalError("Cannot add a finalizer after deletion has begun")
        obj.metadata.finalizers = [*existing, finalizer]

    return await mutate(resource, change, config=config)

remove_finalizer(resource, finalizer, *, config=None) async

Remove only this finalizer after successful, idempotent external cleanup.

Source code in cloudcoil/controller/_mutations.py
75
76
77
78
79
80
81
82
83
84
85
86
async def remove_finalizer[T: Resource](
    resource: T, finalizer: str, *, config: Config | None = None
) -> T:
    """Remove only this finalizer after successful, idempotent external cleanup."""

    def change(obj: T) -> None:
        assert obj.metadata is not None
        existing = obj.metadata.finalizers or []
        if finalizer in existing:
            obj.metadata.finalizers = [value for value in existing if value != finalizer]

    return await mutate(resource, change, config=config)

Dependency-free JSON Patch calculation and optimistic resource diffs.

diff(before, after)

Diff copies of one fetched resource, guarded by UID and resourceVersion.

Returns [] for no changes. Identity/version changes are rejected. Keep the original snapshot unchanged and edit a deep copy. None-valued model fields are omitted, matching Resource writes; clearing a field generates a remove.

Source code in cloudcoil/patches.py
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
def diff(before: Resource, after: Resource) -> list[dict[str, Any]]:
    """Diff copies of one fetched resource, guarded by UID and resourceVersion.

    Returns [] for no changes. Identity/version changes are rejected. Keep the
    original snapshot unchanged and edit a deep copy. None-valued model fields
    are omitted, matching Resource writes; clearing a field generates a remove.
    """
    if before.gvk() != after.gvk() or (before.name, before.namespace) != (
        after.name,
        after.namespace,
    ):
        raise ValueError("A resource diff cannot change kind, name, or namespace")
    if not before.metadata or not before.metadata.uid or not before.resource_version:
        raise ValueError("An optimistic diff needs a fetched resource with UID and resourceVersion")
    if (
        not after.metadata
        or after.metadata.uid != before.metadata.uid
        or after.resource_version != before.resource_version
    ):
        raise ValueError("A resource diff cannot change UID or resourceVersion")
    patch = json_patch(
        before.model_dump(mode="json", by_alias=True, exclude_none=True),
        after.model_dump(mode="json", by_alias=True, exclude_none=True),
    )
    if not patch:
        return []
    return [
        {"op": "test", "path": "/metadata/uid", "value": before.metadata.uid},
        {"op": "test", "path": "/metadata/resourceVersion", "value": before.resource_version},
        *patch,
    ]

json_patch(before, after)

Calculate RFC 6902 operations for JSON values, preserving explicit nulls.

Object members are diffed recursively. Arrays are replaced as a whole; this does not infer Kubernetes strategic-merge keys. Values are copied into the patch so later mutation of the desired document cannot change the request.

Source code in cloudcoil/patches.py
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
def json_patch(before: Any, after: Any) -> list[dict[str, Any]]:
    """Calculate RFC 6902 operations for JSON values, preserving explicit nulls.

    Object members are diffed recursively. Arrays are replaced as a whole; this
    does not infer Kubernetes strategic-merge keys. Values are copied into the
    patch so later mutation of the desired document cannot change the request.
    """
    # Validate JSON values, including rejecting non-finite numbers.
    json.dumps(before, allow_nan=False)
    json.dumps(after, allow_nan=False)
    patch: list[dict[str, Any]] = []

    def visit(old: Any, new: Any, path: str) -> None:
        if isinstance(old, dict) and isinstance(new, dict):
            for key in sorted(old.keys() | new.keys()):
                pointer = f"{path}/{key.replace('~', '~0').replace('/', '~1')}"
                if key not in new:
                    patch.append({"op": "remove", "path": pointer})
                elif key not in old:
                    patch.append({"op": "add", "path": pointer, "value": deepcopy(new[key])})
                else:
                    visit(old[key], new[key], pointer)
        elif json.dumps(old, sort_keys=True) != json.dumps(new, sort_keys=True):
            patch.append({"op": "replace", "path": path, "value": deepcopy(new)})

    visit(before, after, "")
    return patch