22import hashlib as _hashlib
33import inspect
44import json
5+ from contextlib import contextmanager
56import logging
67import os
78import sys
2930from posthog .tracing ._span import inert_span as _inert_span
3031from posthog .capture_compression import (
3132 CaptureCompression ,
33+ _resolve_capture_ai_compression ,
3234 _resolve_capture_compression ,
3335)
3436from posthog .capture_send import (
9193from posthog .request import (
9294 USER_AGENT as _USER_AGENT ,
9395 APIError ,
96+ DatetimeSerializer as _DatetimeSerializer ,
9497 QuotaLimitError ,
9598 RequestsConnectionError ,
9699 RequestsTimeout ,
@@ -227,6 +230,23 @@ def _get_atexit_deadline() -> float:
227230 return _atexit_deadline
228231
229232
233+ def _positive_config_value (
234+ name : str , value , * , integer : bool = False , maximum : Optional [int ] = None
235+ ):
236+ """Return ``value`` if it is positive and no larger than ``maximum``.
237+
238+ Bad lane config is a programming error, so it raises instead of falling
239+ back to a default.
240+ """
241+ allowed = (int ,) if integer else (int , float )
242+ if isinstance (value , bool ) or not isinstance (value , allowed ) or value <= 0 :
243+ kind = "integer" if integer else "number"
244+ raise ValueError (f"{ name } must be a positive { kind } , got { value !r} " )
245+ if maximum is not None and value > maximum :
246+ raise ValueError (f"{ name } must be at most { maximum } , got { value !r} " )
247+ return value
248+
249+
230250def get_identity_state (passed ) -> tuple [str , bool ]:
231251 """Returns the distinct id to use, and whether this is a personless event or not"""
232252 stringified = stringify_id (passed )
@@ -708,6 +728,10 @@ def __init__(
708728 exception_autocapture_refill_rate = ExceptionCapture .DEFAULT_REFILL_RATE ,
709729 exception_autocapture_refill_interval_seconds = ExceptionCapture .DEFAULT_REFILL_INTERVAL_SECONDS ,
710730 capture_compression : Optional [Union [CaptureCompression , str ]] = None ,
731+ capture_ai_compression : Optional [Union [CaptureCompression , str ]] = None ,
732+ capture_ai_max_queue_size : int = 1000 ,
733+ capture_ai_timeout : float = 30 ,
734+ capture_ai_max_event_bytes : int = AI_MAX_MSG_SIZE ,
711735 secret_key = None ,
712736 metrics : Optional [dict ] = None ,
713737 enable_full_ai_capture = False ,
@@ -726,7 +750,8 @@ def __init__(
726750 the corresponding ingestion host.
727751 debug: Enable verbose SDK logging and re-raise errors from public
728752 API methods.
729- max_queue_size: Maximum number of events buffered before upload.
753+ max_queue_size: Maximum number of analytics events buffered before
754+ upload. AI events use ``capture_ai_max_queue_size``.
730755 send: If False, queueing succeeds but events are not sent.
731756 on_error: Optional callback ``(error, batch)`` invoked when an upload
732757 fails: by background consumers, or on the calling thread in
@@ -843,6 +868,21 @@ def __init__(
843868 strings ``"gzip"``/``"deflate"``). When omitted, the
844869 ``POSTHOG_CAPTURE_COMPRESSION`` env var is consulted, then no
845870 compression.
871+ capture_ai_compression: Request-body compression for
872+ ``capture_ai()`` uploads, set independently of
873+ ``capture_compression``. Defaults to no compression, and the
874+ env var does not apply. ``CaptureCompression.ZSTD`` suits large
875+ AI payloads.
876+ capture_ai_max_queue_size: Maximum number of AI events buffered
877+ before upload. Defaults to 1000, lower than ``max_queue_size``
878+ because AI events are much larger.
879+ capture_ai_timeout: Seconds allowed for one AI upload request.
880+ Defaults to 30, longer than ``timeout`` because AI batches are
881+ much larger.
882+ capture_ai_max_event_bytes: Largest serialized AI event the SDK
883+ sends; a larger one is dropped with an error log. Defaults to
884+ the AI endpoint's ceiling plus envelope headroom, and may only
885+ be lowered.
846886
847887 Examples:
848888 ```python
@@ -946,6 +986,21 @@ def __init__(
946986 self ._library_version = VERSION
947987 self ._sdk_info = f"{ self ._library_id } /{ self ._library_version } "
948988 self .capture_compression = _resolve_capture_compression (capture_compression )
989+ self .capture_ai_compression = _resolve_capture_ai_compression (
990+ capture_ai_compression
991+ )
992+ capture_ai_max_queue_size = _positive_config_value (
993+ "capture_ai_max_queue_size" , capture_ai_max_queue_size , integer = True
994+ )
995+ capture_ai_timeout = _positive_config_value (
996+ "capture_ai_timeout" , capture_ai_timeout
997+ )
998+ capture_ai_max_event_bytes = _positive_config_value (
999+ "capture_ai_max_event_bytes" ,
1000+ capture_ai_max_event_bytes ,
1001+ integer = True ,
1002+ maximum = AI_MAX_MSG_SIZE ,
1003+ )
9491004 self .super_properties = super_properties
9501005 # Release id from POSTHOG_RELEASE_ID, attached to every event. Resolved
9511006 # here so the env var is read once per client.
@@ -1055,34 +1110,36 @@ def __init__(
10551110 api_key = self .api_key ,
10561111 host = self .host ,
10571112 on_error = on_error ,
1058- max_queue_size = max_queue_size ,
10591113 thread_count = thread ,
10601114 send = send ,
10611115 flush_at = flush_at ,
10621116 flush_interval = flush_interval ,
10631117 max_retries = self .max_retries ,
1064- timeout = timeout ,
10651118 historical_migration = historical_migration ,
10661119 sdk_info = self ._sdk_info ,
10671120 )
10681121 self ._analytics_lane = _Lane (
10691122 name = "analytics" ,
10701123 ** lane_defaults ,
1124+ max_queue_size = max_queue_size ,
1125+ timeout = timeout ,
10711126 endpoint = _CAPTURE_V1_PATH ,
10721127 max_msg_size = MAX_MSG_SIZE ,
10731128 capture_compression = self .capture_compression ,
10741129 eager_start = not sync_mode ,
10751130 )
10761131 # The AI lane posts to its own endpoint so multi-MB AI events stay off
1077- # the analytics endpoint's smaller caps. It sends uncompressed. Lazy
1078- # start, so the many clients that never emit AI events pay for no extra
1079- # threads.
1132+ # the analytics endpoint's smaller caps, with its own queue, timeout,
1133+ # size guard and compression. Lazy start, so the many clients that never
1134+ # emit AI events pay for no extra threads.
10801135 self ._ai_lane = _Lane (
10811136 name = "ai" ,
10821137 ** lane_defaults ,
1138+ max_queue_size = capture_ai_max_queue_size ,
1139+ timeout = capture_ai_timeout ,
10831140 endpoint = _CAPTURE_AI_V1_PATH ,
1084- max_msg_size = AI_MAX_MSG_SIZE ,
1085- capture_compression = CaptureCompression . NONE ,
1141+ max_msg_size = capture_ai_max_event_bytes ,
1142+ capture_compression = self . capture_ai_compression ,
10861143 eager_start = False ,
10871144 )
10881145 self ._lanes = [self ._analytics_lane , self ._ai_lane ]
@@ -2435,6 +2492,23 @@ def _enqueue(self, msg, disable_geoip, lane=None, property_allowlist=None):
24352492 if self .sync_mode :
24362493 self .log .debug ("enqueued with blocking %s." , msg ["event" ])
24372494
2495+ try :
2496+ event_size = len (json .dumps (msg , cls = _DatetimeSerializer ).encode ())
2497+ except Exception :
2498+ self .log .error ("Unable to serialize event for sizing, dropping." )
2499+ return None
2500+ if event_size > lane .max_msg_size :
2501+ # Log only name and size: AI events may carry unredacted
2502+ # multimodal payloads that must not leak into logs.
2503+ self .log .error (
2504+ "Event %s (%d bytes) exceeds the %dKiB limit for %s, dropping." ,
2505+ msg ["event" ],
2506+ event_size ,
2507+ lane .max_msg_size // 1024 ,
2508+ lane .endpoint ,
2509+ )
2510+ return None
2511+
24382512 def send_sync () -> None :
24392513 # Sync mode bypasses the lane's queue but keeps its wire config,
24402514 # so AI events still post to the AI endpoint.
@@ -2443,7 +2517,7 @@ def send_sync() -> None:
24432517 self .host ,
24442518 [msg ],
24452519 compression = lane .capture_compression ,
2446- timeout = self .timeout ,
2520+ timeout = lane .timeout ,
24472521 max_retries = self .max_retries ,
24482522 historical_migration = self .historical_migration ,
24492523 sdk_info = self ._sdk_info ,
@@ -2679,13 +2753,14 @@ def flush(self, timeout_seconds: Optional[float] = 10) -> None:
26792753 # Spans drain with events: serverless handlers call flush(), not
26802754 # shutdown(), and leaving spans on their own timer would lose them.
26812755 span_flush = self ._start_span_flush (timeout_seconds )
2682- if timeout_seconds is None :
2683- for lane in self ._lanes :
2684- lane .flush (None )
2685- else :
2686- deadline = time .monotonic () + timeout_seconds
2687- for lane in self ._lanes :
2688- lane .flush (max (0.0 , deadline - time .monotonic ()))
2756+ with self ._drain_lanes_together ():
2757+ if timeout_seconds is None :
2758+ for lane in self ._lanes :
2759+ lane .flush (None )
2760+ else :
2761+ deadline = time .monotonic () + timeout_seconds
2762+ for lane in self ._lanes :
2763+ lane .flush (max (0.0 , deadline - time .monotonic ()))
26892764 if span_flush is not None :
26902765 # The last span request is bounded only by the request
26912766 # timeout, so the wait is not.
@@ -2822,7 +2897,36 @@ def _run_lifecycle_cleanup(
28222897 self .log .exception (log_message )
28232898 errors .append (error )
28242899
2900+ @contextmanager
2901+ def _drain_lanes_together (self ):
2902+ """Signal every lane to drain before waiting on any of them.
2903+
2904+ Lanes then drain in parallel under one budget. Otherwise a lane keeps
2905+ batching on its normal cadence while the client waits on the lane before it.
2906+ """
2907+ signals : list [_DrainSignal ] = []
2908+ try :
2909+ for lane in self ._lanes :
2910+ signal = lane ._drain_signal
2911+ signal .request ()
2912+ signals .append (signal )
2913+ yield
2914+ finally :
2915+ for signal in signals :
2916+ signal .complete ()
2917+
28252918 def _flush_or_discard_queues (self , errors : list [Exception ]) -> None :
2919+ try :
2920+ with self ._drain_lanes_together ():
2921+ self ._flush_or_discard_each_lane (errors )
2922+ return
2923+ except Exception as error :
2924+ self .log .exception ("Failed to signal lane drains during lifecycle cleanup" )
2925+ errors .append (error )
2926+ # Each lane's flush signals its own drain, so lanes still drain one by one.
2927+ self ._flush_or_discard_each_lane (errors )
2928+
2929+ def _flush_or_discard_each_lane (self , errors : list [Exception ]) -> None :
28262930 for lane in self ._lanes :
28272931 try :
28282932 if any (consumer .is_alive () for consumer in lane .consumers ):
0 commit comments