From f0397e73144b665f637c4cc6914153e16b2aa0d2 Mon Sep 17 00:00:00 2001 From: Lan_zhijiang Date: Sat, 3 Oct 2026 17:17:56 +0800 Subject: [PATCH 1/5] =?UTF-8?q?docs:=20=E5=BC=95=E7=94=A8=E8=B7=A8=20Peer?= =?UTF-8?q?=20=E5=8F=AF=E8=A7=82=E6=B5=8B=E6=80=A7=E5=A5=91=E7=BA=A6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/_shared | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/_shared b/docs/_shared index d87049c..a603b03 160000 --- a/docs/_shared +++ b/docs/_shared @@ -1 +1 @@ -Subproject commit d87049c49ba10f0a710022dce23abcd316127b3c +Subproject commit a603b036fd41dae5a471dea24cd7609bb4927bd5 From 0eba26add496bc3aebdede8d1158f8f7469c45e8 Mon Sep 17 00:00:00 2001 From: Lan_zhijiang Date: Sat, 3 Oct 2026 18:26:24 +0800 Subject: [PATCH 2/5] =?UTF-8?q?feat:=20=E5=A2=9E=E5=8A=A0=E9=BB=98?= =?UTF-8?q?=E8=AE=A4=E5=85=B3=E9=97=AD=E7=9A=84=E5=88=86=E5=B8=83=E5=BC=8F?= =?UTF-8?q?=E4=B8=8E=20AI=20=E5=8F=AF=E8=A7=82=E6=B5=8B=E6=80=A7?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .changes/observability-foundation.added.md | 1 + .env.example | 26 + app/business/agent/thread.py | 220 ++++---- .../ai/dialects/alibaba_model_studio.py | 23 + app/business/ai/dialects/openai_compatible.py | 9 + app/business/ai/main.py | 201 +++++--- app/business/ai/telemetry.py | 86 +++ app/business/cron.py | 12 +- app/business/info_base/resolver/tools.py | 9 + app/business/info_base/tools.py | 46 ++ app/business/job.py | 85 ++- app/business/peer/http.py | 102 ++-- app/business/sink/agent_query.py | 23 + app/observability.py | 145 ++++++ app/routes/telemetry.py | 105 ++++ app/schemas/job.py | 16 + app/schemas/observability.py | 37 ++ app/settings.py | 1 + docker-compose.yml | 15 +- docs/40-deployment/README.md | 1 + docs/40-deployment/observability.md | 73 +++ docs/40-deployment/runtime-orchestration.md | 3 + docs/openapi.json | 63 +++ libs/obsrv/main.py | 3 + libs/obsrv/setting.py | 5 + libs/obsrv/telemetry.py | 488 ++++++++++++++++++ migrations/revision-integrity.json | 1 + ...b0c855_add_job_submission_trace_context.py | 57 ++ pdm.lock | 163 +++++- pyproject.toml | 2 + run.py | 6 + scripts/observability.py | 33 ++ 32 files changed, 1823 insertions(+), 237 deletions(-) create mode 100644 .changes/observability-foundation.added.md create mode 100644 app/business/ai/telemetry.py create mode 100644 app/observability.py create mode 100644 app/routes/telemetry.py create mode 100644 app/schemas/observability.py create mode 100644 docs/40-deployment/observability.md create mode 100644 libs/obsrv/telemetry.py create mode 100644 migrations/versions/3d9593b0c855_add_job_submission_trace_context.py create mode 100644 scripts/observability.py diff --git a/.changes/observability-foundation.added.md b/.changes/observability-foundation.added.md new file mode 100644 index 0000000..16bde3c --- /dev/null +++ b/.changes/observability-foundation.added.md @@ -0,0 +1 @@ +新增默认关闭的标准 OTLP 可观测性:保留 PostgreSQL 日志,关联跨 Peer Job、AI 调用与结果来源,并提供浏览器认证转发。 diff --git a/.env.example b/.env.example index 8dc9c1a..3180f29 100644 --- a/.env.example +++ b/.env.example @@ -40,3 +40,29 @@ INKCRE_COMPOSE_PROJECT_NAME=inkcre # Flags SKIP_EXTENSIONS_SYNC=0 + +# Optional metadata-only OTLP export. Endpoints/credentials never enable it alone. +# Existing PostgreSQL logging is independent and remains enabled by default. +OBSRV__TELEMETRY_ENABLED=false +# Full per-signal HTTP(S) URLs, including the ingest path. Leave unused signals empty. +# URLs must not contain credentials, query strings or fragments. +OTEL_EXPORTER_OTLP_TRACES_ENDPOINT= +OTEL_EXPORTER_OTLP_LOGS_ENDPOINT= +OTEL_EXPORTER_OTLP_METRICS_ENDPOINT= +# Private server-side ingest headers, using standard OTEL key=value syntax. +# A per-signal value takes precedence over the common value. +OTEL_EXPORTER_OTLP_HEADERS= +OTEL_EXPORTER_OTLP_TRACES_HEADERS= +OTEL_EXPORTER_OTLP_LOGS_HEADERS= +OTEL_EXPORTER_OTLP_METRICS_HEADERS= + +# SDK queue/export self-metrics. Compose forwards this into the process. +# For native launch, export this variable in the process environment as well. +# It does not enable application telemetry by itself. +OTEL_PYTHON_SDK_INTERNAL_METRICS_ENABLED=true + +# Seconds: per-signal timeout overrides this default; accepted range is (0, 30]. +OTEL_EXPORTER_OTLP_TIMEOUT=10 +OTEL_EXPORTER_OTLP_TRACES_TIMEOUT= +OTEL_EXPORTER_OTLP_LOGS_TIMEOUT= +OTEL_EXPORTER_OTLP_METRICS_TIMEOUT= diff --git a/app/business/agent/thread.py b/app/business/agent/thread.py index 5482c06..d8d06c9 100644 --- a/app/business/agent/thread.py +++ b/app/business/agent/thread.py @@ -5,6 +5,7 @@ import asyncio from enum import StrEnum import inspect +import hashlib import json import time import traceback @@ -26,6 +27,7 @@ from .persistence import ThreadID, ThreadPersistenceBackend, ThreadState from .debug import trace from libs.obsrv.main import get_logger +from libs.obsrv.telemetry import operation logger = get_logger().getChild("agent.thread") @@ -92,49 +94,61 @@ def start_turn(self, input: UserMessage) -> asyncio.Task[TurnTermination]: async def _run_turn(self, input: UserMessage) -> TurnTermination: self._turn_index += 1 self._model_calls = 0 - started = time.monotonic() - await trace( - "agent.turn.started", - self.id, - turn=self._turn_index, - input=input, - model=self.model, - max_model_calls=self.max_model_calls_per_turn, - ) - try: - outcome = await self._execute_turn(input) - except asyncio.CancelledError: + with operation( + "agent.turn", + attributes={ + "gen_ai.conversation.id": str(self.id), + "inkcre.agent.turn": self._turn_index, + }, + ) as observation: + span = observation.span + started = time.monotonic() await trace( - "agent.turn.finished", + "agent.turn.started", self.id, turn=self._turn_index, - model_calls=self._model_calls, - outcome="cancelled", - elapsed_seconds=time.monotonic() - started, + input=input, + model=self.model, + max_model_calls=self.max_model_calls_per_turn, ) - raise - except Exception as error: + try: + outcome = await self._execute_turn(input) + except asyncio.CancelledError: + span.set_attribute("inkcre.agent.outcome", "cancelled") + await trace( + "agent.turn.finished", + self.id, + turn=self._turn_index, + model_calls=self._model_calls, + outcome="cancelled", + elapsed_seconds=time.monotonic() - started, + ) + raise + except Exception as error: + span.set_attribute("inkcre.agent.outcome", "failed") + await trace( + "agent.turn.finished", + self.id, + turn=self._turn_index, + model_calls=self._model_calls, + outcome="failed", + error_type=type(error).__name__, + error=str(error), + traceback=traceback.format_exc(), + elapsed_seconds=time.monotonic() - started, + ) + raise await trace( "agent.turn.finished", self.id, turn=self._turn_index, model_calls=self._model_calls, - outcome="failed", - error_type=type(error).__name__, - error=str(error), - traceback=traceback.format_exc(), + outcome=outcome, elapsed_seconds=time.monotonic() - started, ) - raise - await trace( - "agent.turn.finished", - self.id, - turn=self._turn_index, - model_calls=self._model_calls, - outcome=outcome, - elapsed_seconds=time.monotonic() - started, - ) - return outcome + span.set_attribute("inkcre.agent.outcome", outcome.value) + span.set_attribute("inkcre.agent.model_calls", self._model_calls) + return outcome async def _execute_turn(self, input: UserMessage) -> TurnTermination: self._state = await self._persistence.discard_trailing_incomplete_tool_calls(self.id) @@ -142,38 +156,46 @@ async def _execute_turn(self, input: UserMessage) -> TurnTermination: while True: self._model_calls += 1 - await trace( - "agent.model.started", - self.id, - turn=self._turn_index, - call=self._model_calls, - ) - started = time.monotonic() - assistant = await AIManager.chat( - self._state.model, - self._state.messages, - self._state.tools, - self._state.tool_choice, - ) - await trace( - "agent.model.completed", - self.id, - turn=self._turn_index, - call=self._model_calls, - response=assistant, - elapsed_seconds=time.monotonic() - started, - ) - if not assistant.tool_calls: - self._state = await self._persistence.append(self.id, (assistant,)) - return TurnTermination.COMPLETED + with operation( + "agent.step", + attributes={ + "gen_ai.conversation.id": str(self.id), + "inkcre.agent.turn": self._turn_index, + "inkcre.agent.model_call": self._model_calls, + }, + ): + await trace( + "agent.model.started", + self.id, + turn=self._turn_index, + call=self._model_calls, + ) + started = time.monotonic() + assistant = await AIManager.chat( + self._state.model, + self._state.messages, + self._state.tools, + self._state.tool_choice, + ) + await trace( + "agent.model.completed", + self.id, + turn=self._turn_index, + call=self._model_calls, + response=assistant, + elapsed_seconds=time.monotonic() - started, + ) + if not assistant.tool_calls: + self._state = await self._persistence.append(self.id, (assistant,)) + return TurnTermination.COMPLETED - results = await self._execute_tool_batch(assistant) - self._state = await self._persistence.append( - self.id, - (assistant, ToolResultMessage(results=results)), - ) - if self._model_calls >= self._state.max_model_calls_per_turn: - return TurnTermination.MAX_MODEL_CALLS + results = await self._execute_tool_batch(assistant) + self._state = await self._persistence.append( + self.id, + (assistant, ToolResultMessage(results=results)), + ) + if self._model_calls >= self._state.max_model_calls_per_turn: + return TurnTermination.MAX_MODEL_CALLS async def _execute_tool_batch( self, @@ -189,37 +211,61 @@ async def _execute_tool_batch( return tuple(await asyncio.gather(*tasks)) async def _execute_tool_call(self, call: ToolCall) -> ToolResult: - await trace( - "agent.tool.started", - self.id, - turn=self._turn_index, - call=self._model_calls, - tool_call=call, - ) - started = time.monotonic() - try: - result = await self._invoke_tool_call(call) - except asyncio.CancelledError: + with operation( + "agent.tool", + attributes={ + "gen_ai.operation.name": "execute_tool", + "gen_ai.conversation.id": str(self.id), + "inkcre.agent.turn": self._turn_index, + "inkcre.agent.model_call": self._model_calls, + # Provider-generated IDs may contain content; retain correlation as a digest. + "inkcre.agent.tool_call.id_hash": hashlib.sha256( + call.id.encode(errors="replace") + ).hexdigest(), + }, + ) as observation: + span = observation.span + tool = self._tools.get(call.tool) + if tool is not None: + span.set_attribute("gen_ai.tool.name", tool.definition.id) await trace( - "agent.tool.cancelled", + "agent.tool.started", + self.id, + turn=self._turn_index, + call=self._model_calls, + tool_call=call, + ) + started = time.monotonic() + try: + result = await self._invoke_tool_call(call) + except asyncio.CancelledError: + span.set_attribute("inkcre.agent.outcome", "cancelled") + await trace( + "agent.tool.cancelled", + self.id, + turn=self._turn_index, + call=self._model_calls, + tool_call_id=call.id, + tool=call.tool, + elapsed_seconds=time.monotonic() - started, + ) + raise + await trace( + "agent.tool.completed", self.id, turn=self._turn_index, call=self._model_calls, - tool_call_id=call.id, tool=call.tool, + result=result, elapsed_seconds=time.monotonic() - started, ) - raise - await trace( - "agent.tool.completed", - self.id, - turn=self._turn_index, - call=self._model_calls, - tool=call.tool, - result=result, - elapsed_seconds=time.monotonic() - started, - ) - return result + span.set_attribute( + "inkcre.agent.outcome", "error" if result.is_error else "completed" + ) + if result.is_error: + span.set_attribute("error.type", "ToolResultError") + observation.outcome = "error" + return result async def _invoke_tool_call(self, call: ToolCall) -> ToolResult: tool = self._tools.get(call.tool) diff --git a/app/business/ai/dialects/alibaba_model_studio.py b/app/business/ai/dialects/alibaba_model_studio.py index 1227292..ee31b56 100644 --- a/app/business/ai/dialects/alibaba_model_studio.py +++ b/app/business/ai/dialects/alibaba_model_studio.py @@ -3,9 +3,12 @@ from collections.abc import Sequence import json import typing +import time from openai import AsyncStream from openai.types.chat import ChatCompletionChunk +from opentelemetry import trace +from opentelemetry.trace import INVALID_SPAN import pydantic from app.schemas.ai import ( @@ -21,8 +24,11 @@ VideoContentPart, ) +from libs.obsrv.telemetry import is_enabled + from ..contracts import AIOutputContractError from ..main import AIManager +from ..telemetry import record_response_usage from .openai_compatible import OpenAICompatibleConfig, OpenAICompatibleDialect @@ -109,6 +115,12 @@ async def chat( if tool_choice is not None: arguments["tool_choice"] = self._tool_choice_param(tool_choice) + span = trace.get_current_span() if is_enabled() else INVALID_SPAN + span.set_attribute("gen_ai.request.stream", True) + started = time.monotonic() + usage = None + finish_reasons: list[str] = [] + first_chunk = True try: stream = typing.cast( AsyncStream[ChatCompletionChunk], @@ -117,8 +129,18 @@ async def chat( text_parts: list[str] = [] calls: dict[int, dict[str, str]] = {} async for chunk in stream: + if first_chunk: + span.set_attribute( + "gen_ai.response.time_to_first_chunk", time.monotonic() - started + ) + first_chunk = False + # Usage-only terminal chunks have no choices. Totals replace earlier totals. + if chunk.usage is not None: + usage = chunk.usage if not chunk.choices: continue + if chunk.choices[0].finish_reason is not None: + finish_reasons = [chunk.choices[0].finish_reason] delta = chunk.choices[0].delta if isinstance(delta.content, str): text_parts.append(delta.content) @@ -132,6 +154,7 @@ async def chat( if call.function.arguments: current["arguments"] += call.function.arguments finally: + record_response_usage("chat", usage, finish_reasons) await client.close() parsed_calls: list[ToolCall] = [] diff --git a/app/business/ai/dialects/openai_compatible.py b/app/business/ai/dialects/openai_compatible.py index 7dd45c6..8bb24ba 100644 --- a/app/business/ai/dialects/openai_compatible.py +++ b/app/business/ai/dialects/openai_compatible.py @@ -28,6 +28,7 @@ from ..contracts import AIDialectAdapter, AIInputUnavailableError, AIOutputContractError from ..main import AIManager +from ..telemetry import record_response_usage INLINE_MEDIA_MAX_BYTES = 7 * 1024 * 1024 @@ -80,6 +81,7 @@ async def embed( dimensions: int, ) -> tuple[Vector, ...]: client = self._client_factory(self._config(config)) + usage = None try: response = await client.embeddings.create( model=native_model_id, @@ -87,7 +89,9 @@ async def embed( dimensions=dimensions, encoding_format="float", ) + usage = getattr(response, "usage", None) finally: + record_response_usage("embeddings", usage) await client.close() by_index = {item.index: tuple(item.embedding) for item in response.data} @@ -301,12 +305,17 @@ async def chat( if tool_choice is not None: arguments["tool_choice"] = self._tool_choice_param(tool_choice) + usage = None + finish_reasons: list[str] = [] try: response = typing.cast( ChatCompletion, await client.chat.completions.create(**arguments), ) + usage = response.usage + finish_reasons = [choice.finish_reason for choice in response.choices] finally: + record_response_usage("chat", usage, finish_reasons) await client.close() if not response.choices: diff --git a/app/business/ai/main.py b/app/business/ai/main.py index 3fe9dd0..afe888c 100644 --- a/app/business/ai/main.py +++ b/app/business/ai/main.py @@ -6,6 +6,7 @@ import typing import pydantic +from opentelemetry.trace import SpanKind from app.database_contract.profile import BUILTIN_AI_DIALECTS_BY_ID from app.schemas.ai import ( @@ -29,6 +30,9 @@ from app.schemas.info_base.main import Vector from app.persistence.ai.uow import ai_uow +from libs.obsrv.telemetry import operation + +from .telemetry import model_attributes from .contracts import ( AICapabilityUnavailableError, @@ -258,38 +262,54 @@ async def embed( dimensions: int, ) -> tuple[Vector, ...]: """Embed one ordered non-empty text batch through the selected model.""" - if not inputs: - raise ValueError("embedding inputs must not be empty") - if dimensions <= 0: - raise ValueError("embedding dimensions must be positive") - target = await cls._load_target_async(model) - capability = cls._capability(target.model, "embedding") - cls._require_modalities( - target.model, - capability, - input_="text", - output="vector", - ) - vectors = await target.adapter.embed( - target.config, - target.model.native_model_id, - tuple(inputs), - dimensions, - ) - if len(vectors) != len(inputs): - raise AIOutputContractError( - f"Embedding result count {len(vectors)} does not match input count {len(inputs)}" + with operation( + "ai.embed", + kind=SpanKind.CLIENT, + attributes={ + "gen_ai.operation.name": "embeddings", + "inkcre.ai.model.id": model, + "inkcre.ai.usage.input.source": "unavailable", + "inkcre.ai.usage.output.source": "unavailable", + }, + ) as observation: + span = observation.span + if not inputs: + raise ValueError("embedding inputs must not be empty") + if dimensions <= 0: + raise ValueError("embedding dimensions must be positive") + target = await cls._load_target_async(model) + span.set_attributes( + model_attributes( + target.model.native_model_id, target.provider.dialect, target.model.provider + ) + ) + capability = cls._capability(target.model, "embedding") + cls._require_modalities( + target.model, + capability, + input_="text", + output="vector", + ) + vectors = await target.adapter.embed( + target.config, + target.model.native_model_id, + tuple(inputs), + dimensions, ) - for vector in vectors: - if len(vector) != dimensions: + if len(vectors) != len(inputs): raise AIOutputContractError( - f"Embedding dimension {len(vector)} does not match requested {dimensions}" + f"Embedding result count {len(vectors)} does not match input count {len(inputs)}" ) - if not vector or not all(math.isfinite(value) for value in vector): - raise AIOutputContractError("Embedding vectors must be finite and non-empty") - if not any(value != 0 for value in vector): - raise AIOutputContractError("Embedding vectors must be non-zero") - return vectors + for vector in vectors: + if len(vector) != dimensions: + raise AIOutputContractError( + f"Embedding dimension {len(vector)} does not match requested {dimensions}" + ) + if not vector or not all(math.isfinite(value) for value in vector): + raise AIOutputContractError("Embedding vectors must be finite and non-empty") + if not any(value != 0 for value in vector): + raise AIOutputContractError("Embedding vectors must be non-zero") + return vectors @classmethod async def chat( @@ -300,60 +320,77 @@ async def chat( tool_choice: ToolChoice | None = None, ) -> AssistantMessage: """Execute one provider-neutral chat model call without owning history.""" - history = validate_message_history(messages) - target = await cls._load_target_async(model) - capability = cls._capability(target.model, "chat") - # System instructions, Tool schemas/results and Assistant history all travel - # through the textual chat channel even when the latest User turn is media-only. - input_modalities: set[str] = {"text"} - for message in history: - if not isinstance(message, UserMessage): - continue - for part in message.content: - if isinstance(part, TextContentPart): - input_modalities.add("text") - elif isinstance(part, ImageContentPart): - input_modalities.add("image") - elif isinstance(part, AudioContentPart): - input_modalities.add("audio") - elif isinstance(part, VideoContentPart): - input_modalities.add("video") - for modality in sorted(input_modalities): - cls._require_modalities( - target.model, - capability, - input_=modality, - output="text", + with operation( + "ai.chat", + kind=SpanKind.CLIENT, + attributes={ + "gen_ai.operation.name": "chat", + "inkcre.ai.model.id": model, + "inkcre.ai.usage.input.source": "unavailable", + "inkcre.ai.usage.output.source": "unavailable", + }, + ) as observation: + span = observation.span + history = validate_message_history(messages) + target = await cls._load_target_async(model) + span.set_attributes( + model_attributes( + target.model.native_model_id, target.provider.dialect, target.model.provider + ) ) - if not target.adapter.supports_input_modality("chat", modality): - raise AIInputUnavailableError( - f"AI dialect {target.provider.dialect!r} cannot convey chat modality {modality!r}" + capability = cls._capability(target.model, "chat") + # System instructions, Tool schemas/results and Assistant history all travel + # through the textual chat channel even when the latest User turn is media-only. + input_modalities: set[str] = {"text"} + for message in history: + if not isinstance(message, UserMessage): + continue + for part in message.content: + if isinstance(part, TextContentPart): + input_modalities.add("text") + elif isinstance(part, ImageContentPart): + input_modalities.add("image") + elif isinstance(part, AudioContentPart): + input_modalities.add("audio") + elif isinstance(part, VideoContentPart): + input_modalities.add("video") + for modality in sorted(input_modalities): + cls._require_modalities( + target.model, + capability, + input_=modality, + output="text", ) + if not target.adapter.supports_input_modality("chat", modality): + raise AIInputUnavailableError( + f"AI dialect {target.provider.dialect!r} cannot convey " + f"chat modality {modality!r}" + ) - tool_ids = tuple(tool.id for tool in tools) - if len(tool_ids) != len(set(tool_ids)): - raise ValueError("Tool IDs must be unique") - requires_tool_calling = bool(tools) or tool_choice is not None - if requires_tool_calling: - if "tool_calling" not in capability.features or not target.adapter.supports_feature( - "chat", "tool_calling" - ): - raise AIFeatureUnavailableError( - f"AI model {model} and dialect {target.provider.dialect!r} do not jointly " - "support tool_calling" - ) - if tool_choice is not None: - if not tools: - raise ValueError("tool_choice requires at least one Tool") - if not target.adapter.supports_tool_choice(tool_choice): - raise AIFeatureUnavailableError( - f"AI dialect {target.provider.dialect!r} cannot represent tool_choice" - ) + tool_ids = tuple(tool.id for tool in tools) + if len(tool_ids) != len(set(tool_ids)): + raise ValueError("Tool IDs must be unique") + requires_tool_calling = bool(tools) or tool_choice is not None + if requires_tool_calling: + if "tool_calling" not in capability.features or not target.adapter.supports_feature( + "chat", "tool_calling" + ): + raise AIFeatureUnavailableError( + f"AI model {model} and dialect {target.provider.dialect!r} do not jointly " + "support tool_calling" + ) + if tool_choice is not None: + if not tools: + raise ValueError("tool_choice requires at least one Tool") + if not target.adapter.supports_tool_choice(tool_choice): + raise AIFeatureUnavailableError( + f"AI dialect {target.provider.dialect!r} cannot represent tool_choice" + ) - return await target.adapter.chat( - target.config, - target.model.native_model_id, - history, - tuple(tools), - tool_choice, - ) + return await target.adapter.chat( + target.config, + target.model.native_model_id, + history, + tuple(tools), + tool_choice, + ) diff --git a/app/business/ai/telemetry.py b/app/business/ai/telemetry.py new file mode 100644 index 0000000..00431aa --- /dev/null +++ b/app/business/ai/telemetry.py @@ -0,0 +1,86 @@ +"""Content-free mapping for the pinned GenAI semantic conventions. + +Keys follow semantic-conventions-genai e07f4ebacb08f56db8c4c882d117720333fbca04. +Provider responses are untrusted: copy numeric usage and known finish reasons only. +""" + +from collections.abc import Sequence +import re + +from openai.types.completion_usage import CompletionUsage +from openai.types.create_embedding_response import Usage as EmbeddingUsage +from opentelemetry import trace +from opentelemetry.util.types import AttributeValue + +from libs.obsrv.telemetry import is_enabled, record_ai_usage + + +_IDENTIFIER = re.compile(r"[A-Za-z0-9][A-Za-z0-9._:/-]{0,127}\Z") +_FINISH_REASONS = frozenset( + {"stop", "length", "tool_calls", "content_filter", "function_call"} +) + + +def model_attributes( + native_model_id: str, dialect: str, provider_id: int +) -> dict[str, AttributeValue]: + attributes: dict[str, AttributeValue] = { + "inkcre.ai.provider.id": provider_id, + "inkcre.ai.dialect": dialect if _IDENTIFIER.fullmatch(dialect) else "other", + "gen_ai.provider.name": { + "core.openai-compatible.v1": "openai", + "core.alibaba-model-studio.v1": "alibaba", + }.get(dialect, "other"), + } + # This is the configured model identifier, never the provider's arbitrary echo. + if _IDENTIFIER.fullmatch(native_model_id): + attributes["gen_ai.request.model"] = native_model_id + return attributes + + +def record_response_usage( + operation: str, + usage: CompletionUsage | EmbeddingUsage | None, + finish_reasons: Sequence[str] = (), +) -> None: + """Record one response's totals once; absent or invalid totals stay unknown.""" + if not is_enabled(): + return + span = trace.get_current_span() + input_tokens = getattr(usage, "prompt_tokens", None) + output_tokens = getattr(usage, "completion_tokens", None) + input_tokens = ( + input_tokens if type(input_tokens) is int and 0 <= input_tokens < 2**63 else None + ) + output_tokens = ( + output_tokens if type(output_tokens) is int and 0 <= output_tokens < 2**63 else None + ) + for direction, value in (("input", input_tokens), ("output", output_tokens)): + span.set_attribute( + f"inkcre.ai.usage.{direction}.source", + "provider" if value is not None else "unavailable", + ) + if value is not None: + span.set_attribute(f"gen_ai.usage.{direction}_tokens", value) + if isinstance(usage, CompletionUsage): + for key, value in ( + ( + "gen_ai.usage.cache_read.input_tokens", + getattr(usage.prompt_tokens_details, "cached_tokens", None), + ), + ( + "gen_ai.usage.reasoning.output_tokens", + getattr(usage.completion_tokens_details, "reasoning_tokens", None), + ), + ): + if type(value) is int and 0 <= value < 2**63: + span.set_attribute(key, value) + if finish_reasons: + span.set_attribute( + "gen_ai.response.finish_reasons", + tuple( + reason if isinstance(reason, str) and reason in _FINISH_REASONS else "other" + for reason in finish_reasons[:16] + ), + ) + record_ai_usage(operation, input_tokens=input_tokens, output_tokens=output_tokens) diff --git a/app/business/cron.py b/app/business/cron.py index 5d9b171..3bb7765 100644 --- a/app/business/cron.py +++ b/app/business/cron.py @@ -15,6 +15,7 @@ from app.schemas.cron import CronForm, CronID, CronModel, CronUpdateForm from app.schemas.job import JobModel, JobStatus from libs.obsrv.main import get_logger +from libs.obsrv.telemetry import operation, emit_event from app.validation import input_path @@ -87,6 +88,11 @@ async def check(cls) -> int: @classmethod async def _materialize(cls, cron_id: CronID, timezone: ZoneInfo) -> int: + with operation("cron.materialize", attributes={"inkcre.cron.id": cron_id}): + return await cls._materialize_occurrence(cron_id, timezone) + + @classmethod + async def _materialize_occurrence(cls, cron_id: CronID, timezone: ZoneInfo) -> int: async with cron_uow() as uow: cron = await uow.crons.get(cron_id, lock=True, skip_locked=True) if cron is None or not cron.enabled: @@ -117,7 +123,8 @@ async def _materialize(cls, cron_id: CronID, timezone: ZoneInfo) -> int: cron.last_job = job.id cron.last_scheduled_for = occurrence await uow.crons.save(cron) - return 1 + emit_event("job.submitted", {"inkcre.job.id": job.id or 0, "inkcre.cron.id": cron_id}) + return 1 @classmethod async def run_now(cls, cron_id: CronID) -> JobModel: @@ -132,7 +139,8 @@ async def run_now(cls, cron_id: CronID) -> JobModel: cron.job_timeout_seconds, uow=uow, ) - return job + emit_event("job.submitted", {"inkcre.job.id": job.id or 0, "inkcre.cron.id": cron_id}) + return job @classmethod async def create(cls, form: CronForm) -> CronModel: diff --git a/app/business/info_base/resolver/tools.py b/app/business/info_base/resolver/tools.py index b409bd1..2209197 100644 --- a/app/business/info_base/resolver/tools.py +++ b/app/business/info_base/resolver/tools.py @@ -9,6 +9,7 @@ from app.business.info_base.services import BlockService from app.schemas.ai import JSONValue from app.schemas.info_base.block import BlockID, ResolverType +from libs.obsrv.telemetry import emit_event from .main import ResolverManager @@ -257,6 +258,14 @@ async def resolver(input: ResolverMetaToolInput) -> JSONValue: } ) else: + emit_event( + "inkcre.entity.read", + { + "inkcre.entity.read.kind": "resolved", + "inkcre.entity.block_ids": (call.block_id,), + "inkcre.entity.read.count": 1, + }, + ) results.append( { "index": index, diff --git a/app/business/info_base/tools.py b/app/business/info_base/tools.py index a469679..a1dbdb0 100644 --- a/app/business/info_base/tools.py +++ b/app/business/info_base/tools.py @@ -7,6 +7,10 @@ from app.business.agent import AgentManager from app.business.agent.projection import project_json from app.schemas.ai import JSONValue +from app.schemas.info_base.block import BlockModel +from app.schemas.info_base.relation import RelationModel +from app.schemas.lexical_retrieval import LexicalRetrievalResult +from libs.obsrv.telemetry import emit_event from .main import InfoBaseManager @@ -67,6 +71,31 @@ async def retrieve(input: RetrieveInput) -> JSONValue: continue result = branch.result assert result is not None + if isinstance(result, LexicalRetrievalResult): + block_ids = tuple( + match.block.id for match in result.matches if match.block.id is not None + ) + relation_ids: tuple[int, ...] = () + else: + block_ids = tuple( + match.entity.id + for match in result.matches + if match.type == "block" and match.entity.id is not None + ) + relation_ids = tuple( + match.entity.id + for match in result.matches + if match.type == "relation" and match.entity.id is not None + ) + emit_event( + "inkcre.retrieval.candidates", + { + "inkcre.retrieval.method": "lexical" if name == "lexical" else "semantic", + "inkcre.entity.block_ids": block_ids, + "inkcre.entity.relation_ids": relation_ids, + "inkcre.retrieval.candidate_count": len(result.matches), + }, + ) if name == "lexical": payload[name] = { "matches": [ @@ -103,4 +132,21 @@ async def get_entities(input: GetEntitiesInput) -> JSONValue: tuple((reference.type, reference.id) for reference in input.entities), random_count=input.random_count, ) + emit_event( + "inkcre.entity.read", + { + "inkcre.entity.read.kind": "persisted", + "inkcre.entity.block_ids": tuple( + entity.id + for entity in result + if isinstance(entity, BlockModel) and entity.id is not None + ), + "inkcre.entity.relation_ids": tuple( + entity.id + for entity in result + if isinstance(entity, RelationModel) and entity.id is not None + ), + "inkcre.entity.read.count": sum(entity is not None for entity in result), + }, + ) return project_json(result) diff --git a/app/business/job.py b/app/business/job.py index fb27053..a85d1bb 100644 --- a/app/business/job.py +++ b/app/business/job.py @@ -13,6 +13,9 @@ from app.scheduler import scheduler, with_trace_id from app.schemas.job import JobID, JobModel, JobStatus, JobTypeID, JobTypeModel from libs.obsrv.main import get_logger +from libs.obsrv.telemetry import emit_event, operation +from app.observability import submission_context, submission_links +from opentelemetry.context import Context LOGGER = get_logger().getChild(__name__) @@ -161,7 +164,9 @@ async def create( ) -> JobModel: """Validate and persist one independent pending Job.""" async with job_uow() as uow: - return await cls.create_in_uow(job_type, parameters, timeout_seconds, uow=uow) + job = await cls.create_in_uow(job_type, parameters, timeout_seconds, uow=uow) + emit_event("job.submitted", {"inkcre.job.id": job.id or 0}) + return job @classmethod async def create_in_uow( @@ -182,12 +187,21 @@ async def create_in_uow( if effective_timeout <= 0: raise ValueError("Job timeout_seconds must be positive") - job = JobModel( - type=job_type, - parameters=normalized, - timeout_seconds=effective_timeout, - ) - return await uow.jobs.create(job) + with operation("job.submit") as observation: + span = observation.span + carrier = submission_context() + job = JobModel( + type=job_type, + parameters=normalized, + timeout_seconds=effective_timeout, + submission_traceparent=carrier["submission_traceparent"], + submission_tracestate=carrier["submission_tracestate"], + ) + job = await uow.jobs.create(job) + span.set_attribute("inkcre.job.id", job.id or 0) + # The composing use case still owns commit; this span only proves the flush. + span.set_attribute("inkcre.job.write_phase", "flushed") + return job @classmethod async def _prepare( @@ -289,7 +303,12 @@ async def _close(cls, job: JobModel, status: JobStatus) -> bool: # Cleanup is independent of handler cancellation and cannot wait indefinitely. async with asyncio.timeout(10): async with job_uow() as uow: - return await uow.jobs.close(job, status) + closed = await uow.jobs.close(job, status) + if closed: + emit_event( + "job.closed", {"inkcre.job.id": job.id or 0, "inkcre.job.status": status.value} + ) + return closed @classmethod async def run(cls, job_id: JobID) -> bool: @@ -321,23 +340,39 @@ async def run(cls, job_id: JobID) -> bool: task = asyncio.current_task() assert task is not None cls._active[job_id] = _Execution(task) - try: - async with asyncio.timeout(claimed.timeout_seconds): - await handler.handle(claimed, parameters) - except asyncio.CancelledError: - LOGGER.info("Job execution aborted", extra={"job_id": job_id}) - await cls._close(claimed, JobStatus.ABORTED) - except TimeoutError: - LOGGER.warning("Job execution timed out", extra={"job_id": job_id}) - await cls._close(claimed, JobStatus.TIMED_OUT) - except Exception as error: - LOGGER.exception("Job execution failed", extra={"job_id": job_id}) - claimed.state = {**claimed.state, "error": str(error)} - await cls._close(claimed, JobStatus.FAILED) - else: - await cls._close(claimed, JobStatus.FINISHED) - finally: - cls._active.pop(job_id, None) + with operation( + "job.execute", + attributes={"inkcre.job.id": job_id}, + context=Context(), + links=submission_links(claimed.submission_traceparent, claimed.submission_tracestate), + ) as observation: + span = observation.span + emit_event("job.started", {"inkcre.job.id": job_id}) + try: + async with asyncio.timeout(claimed.timeout_seconds): + await handler.handle(claimed, parameters) + except asyncio.CancelledError: + LOGGER.info("Job execution aborted", extra={"job_id": job_id}) + observation.outcome = "cancelled" + span.set_attribute("inkcre.job.status", "aborted") + await cls._close(claimed, JobStatus.ABORTED) + except TimeoutError: + LOGGER.warning("Job execution timed out", extra={"job_id": job_id}) + observation.outcome = "error" + span.set_attribute("inkcre.job.status", "timed_out") + await cls._close(claimed, JobStatus.TIMED_OUT) + except Exception as error: + LOGGER.exception("Job execution failed", extra={"job_id": job_id}) + observation.outcome = "error" + span.set_attribute("error.type", type(error).__name__) + span.set_attribute("inkcre.job.status", "failed") + claimed.state = {**claimed.state, "error": str(error)} + await cls._close(claimed, JobStatus.FAILED) + else: + span.set_attribute("inkcre.job.status", "finished") + await cls._close(claimed, JobStatus.FINISHED) + finally: + cls._active.pop(job_id, None) return True @classmethod diff --git a/app/business/peer/http.py b/app/business/peer/http.py index 77cc07a..6a3a7c0 100644 --- a/app/business/peer/http.py +++ b/app/business/peer/http.py @@ -26,6 +26,9 @@ PeerProtocolResponse, ) from app.settings import settings +from app.observability import inject_context +from libs.obsrv.telemetry import operation +from opentelemetry.trace import SpanKind from .contracts import ( PeerOutcomeUnknown, @@ -132,41 +135,68 @@ async def execute(self, payload: JSONValue) -> JSONValue: if "body" in request.model_fields_set: kwargs["json"] = request.body - try: - async with httpx.AsyncClient( - timeout=settings.peer_http_timeout_seconds, - ) as client: - response = await client.request(**kwargs) - except (httpx.ConnectError, httpx.ConnectTimeout, httpx.PoolTimeout) as error: - raise PeerRequestNotExecuted(f"Could not dispatch to Peer {self.peer.id}") from error - except httpx.RequestError as error: - raise PeerOutcomeUnknown( - f"Peer {self.peer.id} dispatch outcome is unknown" - ) from error - - if response.headers.get(PEER_EXECUTION_HEADER, "").strip().lower() == PEER_NOT_EXECUTED: - raise PeerRequestNotExecuted( - f"Peer {self.peer.id} reported that execution did not begin" - ) - - grouped_headers: defaultdict[str, list[str]] = defaultdict(list) - for name, value in response.headers.multi_items(): - if name.lower() != PEER_EXECUTION_HEADER.lower(): - grouped_headers[name.lower()].append(value) - - response_fields: dict[str, typing.Any] = { - "status": response.status_code, - "headers": grouped_headers, - } - if response.content: + with operation( + "peer.http", + attributes={"inkcre.peer.target_id": str(self.peer.id)}, + kind=SpanKind.CLIENT, + ) as observation: + span = observation.span + propagated = inject_context() + if propagated: + # The transport owns Trace Context; business payload headers cannot duplicate it. + headers[:] = [ + (key, value) + for key, value in headers + if key.lower() not in {"traceparent", "tracestate"} + ] + headers.extend(propagated.items()) try: - response_fields["body"] = response.json() - except json.JSONDecodeError as error: - raise PeerProtocolError( - f"Peer {self.peer.id} returned a non-JSON HTTP body" + async with httpx.AsyncClient( + timeout=settings.peer_http_timeout_seconds, + ) as client: + response = await client.request(**kwargs) + except (httpx.ConnectError, httpx.ConnectTimeout, httpx.PoolTimeout) as error: + span.set_attribute("inkcre.peer.outcome", "not_executed") + raise PeerRequestNotExecuted( + f"Could not dispatch to Peer {self.peer.id}" ) from error - normalized = PeerProtocolResponse.model_validate(response_fields) - return typing.cast( - JSONValue, - normalized.model_dump(mode="json", exclude_unset=True), - ) + except httpx.RequestError as error: + span.set_attribute("inkcre.peer.outcome", "unknown") + raise PeerOutcomeUnknown( + f"Peer {self.peer.id} dispatch outcome is unknown" + ) from error + + if ( + response.headers.get(PEER_EXECUTION_HEADER, "").strip().lower() == PEER_NOT_EXECUTED + ): + span.set_attribute("inkcre.peer.outcome", "not_executed") + raise PeerRequestNotExecuted( + f"Peer {self.peer.id} reported that execution did not begin" + ) + + span.set_attribute("inkcre.peer.outcome", "responded") + span.set_attribute("http.response.status_code", response.status_code) + if response.status_code >= 500: + observation.outcome = "error" + + grouped_headers: defaultdict[str, list[str]] = defaultdict(list) + for name, value in response.headers.multi_items(): + if name.lower() != PEER_EXECUTION_HEADER.lower(): + grouped_headers[name.lower()].append(value) + + response_fields: dict[str, typing.Any] = { + "status": response.status_code, + "headers": grouped_headers, + } + if response.content: + try: + response_fields["body"] = response.json() + except json.JSONDecodeError as error: + raise PeerProtocolError( + f"Peer {self.peer.id} returned a non-JSON HTTP body" + ) from error + normalized = PeerProtocolResponse.model_validate(response_fields) + return typing.cast( + JSONValue, + normalized.model_dump(mode="json", exclude_unset=True), + ) diff --git a/app/business/sink/agent_query.py b/app/business/sink/agent_query.py index a9913bc..82e2761 100644 --- a/app/business/sink/agent_query.py +++ b/app/business/sink/agent_query.py @@ -21,6 +21,7 @@ ) from app.schemas.job import JobModel from app.schemas.sink import SinkModel +from libs.obsrv.telemetry import emit_event from .base import SinkBase from .errors import SinkNotFoundError, SinkStateConflictError @@ -153,6 +154,28 @@ async def execute(self, job: JobModel, query: str) -> None: result = _last_query_result(thread.messages) if result is not None: job.state = {**job.state, "result": result} + # Emit only the closed submission selected for this Job, not every attempt. + references = typing.cast(list[dict[str, JSONValue]], result["references"]) + reported = [ + reference + for reference in references[:128] + if type(reference["id"]) is int and 0 < typing.cast(int, reference["id"]) < 2**63 + ] + emit_event( + "inkcre.agent.query.result", + { + "inkcre.job.id": typing.cast(int, job.id), + "gen_ai.conversation.id": str(thread.id), + "inkcre.entity.block_ids": tuple( + typing.cast(int, ref["id"]) for ref in reported if ref["type"] == "block" + ), + "inkcre.entity.relation_ids": tuple( + typing.cast(int, ref["id"]) for ref in reported if ref["type"] == "relation" + ), + "inkcre.agent.reference_count": len(references), + "inkcre.agent.unreported_reference_count": len(references) - len(reported), + }, + ) job.state = {**job.state, "termination": termination.value} if result is None: raise AgentQueryResultMissingError( diff --git a/app/observability.py b/app/observability.py new file mode 100644 index 0000000..d3a2fda --- /dev/null +++ b/app/observability.py @@ -0,0 +1,145 @@ +"""Runtime projection of optional deployment telemetry and context boundaries.""" + +import asyncio +from collections.abc import Mapping + +from opentelemetry.context import Context +from opentelemetry.trace import Link, SpanKind +from opentelemetry.trace.propagation.tracecontext import TraceContextTextMapPropagator +from opentelemetry import trace +from starlette.types import ASGIApp, Message, Receive, Scope, Send + +from app.business.deployment_config import DeploymentConfigManager, DeploymentConfigService +from app.settings import settings +from app.schemas.observability import CONFIG_KEY, CONFIG_SCHEMA, ObservabilityConfig +from app.version import APPLICATION_VERSION +from libs.obsrv.main import get_logger +from libs.obsrv.telemetry import emit_event, is_enabled, operation, start_telemetry + + +PROPAGATOR = TraceContextTextMapPropagator() + + +DeploymentConfigManager.register_schema( + CONFIG_SCHEMA, ObservabilityConfig, keys=(CONFIG_KEY,) +) + + +async def initialize_telemetry() -> None: + """Read once after database admission; telemetry cannot fail business readiness.""" + if not settings.obsrv.telemetry_enabled: + return + attributes: dict[str, str] = {"inkcre.peer.id": str(settings.peer_id)} + endpoints: dict[str, str] = {} + try: + async with asyncio.timeout(2): + config = await DeploymentConfigService.get(CONFIG_KEY) + if not isinstance(config, ObservabilityConfig): + raise ValueError("missing or unexpected observability schema") + attributes["inkcre.deployment.id"] = str(config.deployment_id) + if config.otlp_http_endpoints is not None: + endpoints = config.otlp_http_endpoints.model_dump(mode="json", exclude_none=True) + except Exception as error: + # Configuration may contain credentials despite the admitted schema. Never render it. + get_logger().warning("Telemetry configuration unavailable (%s)", type(error).__name__) + return + try: + start_telemetry( + enabled=True, + service_version=APPLICATION_VERSION, + resource_attributes=attributes, + endpoints=endpoints, + ) + except Exception as error: + get_logger().warning("Telemetry initialization failed (%s)", type(error).__name__) + + +def inject_context() -> dict[str, str]: + """Inject only standard Trace Context when this runtime explicitly enabled telemetry.""" + carrier: dict[str, str] = {} + if is_enabled(): + PROPAGATOR.inject(carrier) + return carrier + + +def submission_context() -> dict[str, str | None]: + carrier = inject_context() + state = carrier.get("tracestate") + if state is not None and len(state.encode("utf-8")) > 512: + state = None + emit_event("telemetry.carrier.dropped", {"reason": "tracestate_capacity"}) + return { + "submission_traceparent": carrier.get("traceparent"), + "submission_tracestate": state, + } + + +def extract_context(carrier: Mapping[str, str]) -> Context: + """Use the SDK parser within the application's bounded, two-header carrier.""" + if not is_enabled(): + return Context() + bounded = { + key: value + for key in ("traceparent", "tracestate") + if (value := carrier.get(key)) is not None and len(value.encode("utf-8")) <= 512 + } + return PROPAGATOR.extract(bounded, context=Context()) + + +def submission_links(parent: str | None, state: str | None) -> tuple[Link, ...]: + carrier = {} + if parent is not None: + carrier["traceparent"] = parent + if state is not None: + carrier["tracestate"] = state + context = trace.get_current_span(extract_context(carrier)).get_span_context() + return (Link(context),) if context.is_valid else () + + +class TelemetryMiddleware: + """Observe complete HTTP responses without recording paths, headers or bodies.""" + + def __init__(self, app: ASGIApp): + self.app = app + + async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None: + if scope["type"] != "http" or not is_enabled(): + await self.app(scope, receive, send) + return + carrier = { + key.decode("ascii"): value.decode("latin-1") + for key, value in scope.get("headers", ()) + if key in {b"traceparent", b"tracestate"} + } + method = scope.get("method", "") + if method not in { + "GET", + "POST", + "PUT", + "PATCH", + "DELETE", + "HEAD", + "OPTIONS", + "CONNECT", + "TRACE", + }: + method = "_OTHER" + with operation( + "http.server", + attributes={"http.request.method": method}, + context=extract_context(carrier), + kind=SpanKind.SERVER, + ) as observation: + span = observation.span + + async def observed_send(message: Message) -> None: + if message["type"] == "http.response.start": + span.set_attribute("http.response.status_code", message["status"]) + route = scope.get("route") + if route is not None and isinstance(getattr(route, "path", None), str): + span.set_attribute("http.route", route.path) + if message["status"] >= 500: + observation.outcome = "error" + await send(message) + + await self.app(scope, receive, observed_send) diff --git a/app/routes/telemetry.py b/app/routes/telemetry.py new file mode 100644 index 0000000..a80e5b6 --- /dev/null +++ b/app/routes/telemetry.py @@ -0,0 +1,105 @@ +"""Authenticated, bounded OTLP forwarding for browser Peers without ingest secrets.""" + +import asyncio +from typing import Literal + +import fastapi +from google.protobuf.message import DecodeError # type: ignore[untyped-import] +import httpx +from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ( + ExportLogsServiceRequest, + ExportLogsServiceResponse, +) +from opentelemetry.proto.collector.metrics.v1.metrics_service_pb2 import ( + ExportMetricsServiceRequest, + ExportMetricsServiceResponse, +) +from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( + ExportTraceServiceRequest, + ExportTraceServiceResponse, +) + +from libs.obsrv.telemetry import get_export_connection + + +ROUTER = fastapi.APIRouter(prefix="/telemetry", tags=["telemetry"]) +_REQUESTS = { + "traces": ExportTraceServiceRequest, + "logs": ExportLogsServiceRequest, + "metrics": ExportMetricsServiceRequest, +} +_RESPONSES = { + "traces": ExportTraceServiceResponse, + "logs": ExportLogsServiceResponse, + "metrics": ExportMetricsServiceResponse, +} +_MAX_BYTES = 256 * 1024 +_MAX_IN_FLIGHT = 4 +_in_flight = 0 + + +@ROUTER.post("/v1/{signal}", response_class=fastapi.Response) +async def export_telemetry( + signal: Literal["traces", "logs", "metrics"], request: fastapi.Request +) -> fastapi.Response: + """Forward standard OTLP using this Peer's enabled private destination. + + The core router requires the existing Peer JWT. The caller cannot choose an + upstream URL or forward headers. No application payload or durable queue is created. + """ + global _in_flight + connection = get_export_connection(signal) + if connection is None: + raise fastapi.HTTPException(503, "Telemetry destination is not enabled") + if _in_flight >= _MAX_IN_FLIGHT: + raise fastapi.HTTPException(429, "Telemetry forwarding capacity reached") + media_type = request.headers.get("content-type", "").partition(";")[0].strip() + if media_type != "application/x-protobuf": + raise fastapi.HTTPException(415, "OTLP protobuf is required") + if request.headers.get("content-encoding", "identity") != "identity": + raise fastapi.HTTPException(415, "Compressed telemetry requests are not supported") + + endpoint, headers, timeout = connection + # No await between admission and increment: the ASGI event loop owns this bound. + _in_flight += 1 + try: + async with asyncio.timeout(timeout + 2): + body = bytearray() + async for chunk in request.stream(): + if len(body) + len(chunk) > _MAX_BYTES: + raise fastapi.HTTPException(413, "Telemetry request exceeds 256 KiB") + body.extend(chunk) + message = _REQUESTS[signal]() + try: + message.ParseFromString(bytes(body)) + except DecodeError as error: + raise fastapi.HTTPException(400, "Invalid OTLP request") from error + + headers = { + **{ + key: value + for key, value in headers.items() + if key.lower() not in {"content-type", "accept"} + }, + "Content-Type": "application/x-protobuf", + "Accept": "application/x-protobuf", + } + async with httpx.AsyncClient(timeout=timeout, follow_redirects=False) as client: + response = await client.post( + endpoint, headers=headers, content=message.SerializeToString() + ) + if response.status_code not in (200, 204): + # An upstream error body could echo server-side authentication or infrastructure. + raise fastapi.HTTPException(502, "Telemetry destination did not accept the batch") + exported = _RESPONSES[signal]() + try: + exported.ParseFromString(response.content) + except DecodeError as error: + raise fastapi.HTTPException( + 502, "Invalid telemetry destination response" + ) from error + return fastapi.Response(exported.SerializeToString(), media_type=media_type) + except (TimeoutError, httpx.HTTPError) as error: + raise fastapi.HTTPException(502, "Telemetry forwarding did not complete") from error + finally: + _in_flight -= 1 diff --git a/app/schemas/job.py b/app/schemas/job.py index 6e7fc6e..740c0a8 100644 --- a/app/schemas/job.py +++ b/app/schemas/job.py @@ -64,6 +64,14 @@ class JobModel(sqlmodel.SQLModel, table=True): __tablename__ = "jobs" # type: ignore __table_args__ = ( sqlalchemy.Index("jobs_status_idx", "status"), + sqlalchemy.CheckConstraint( + "octet_length(submission_traceparent) <= 512", + name="jobs_submission_traceparent_capacity", + ), + sqlalchemy.CheckConstraint( + "octet_length(submission_tracestate) <= 512", + name="jobs_submission_tracestate_capacity", + ), sqlalchemy.CheckConstraint( "timeout_seconds > 0", name="jobs_timeout_seconds_positive", @@ -100,6 +108,14 @@ class JobModel(sqlmodel.SQLModel, table=True): server_default=sqlalchemy.text("'{}'::jsonb"), ), ) + submission_traceparent: str | None = sqlmodel.Field( + default=None, + sa_column=sqlalchemy.Column(sqlalchemy.Text, nullable=True), + ) + submission_tracestate: str | None = sqlmodel.Field( + default=None, + sa_column=sqlalchemy.Column(sqlalchemy.Text, nullable=True), + ) state: dict[str, typing.Any] = sqlmodel.Field( default_factory=dict, sa_column=sqlalchemy.Column( diff --git a/app/schemas/observability.py b/app/schemas/observability.py new file mode 100644 index 0000000..b8e588a --- /dev/null +++ b/app/schemas/observability.py @@ -0,0 +1,37 @@ +"""Non-secret deployment telemetry configuration shared by admitted Peers.""" + +import uuid +import pydantic + +CONFIG_KEY = "inkcre.observability" +CONFIG_SCHEMA = "inkcre.observability.v1" + + +def _credential_free_url(value: pydantic.AnyHttpUrl | None): + if value is not None and ( + value.username or value.password or value.query or value.fragment + ): + raise ValueError("observability URLs cannot contain credentials, queries or fragments") + return value + + +class TelemetryEndpoints(pydantic.BaseModel): + """Complete OTLP/HTTP destinations; authentication stays in local runtime config.""" + + model_config = pydantic.ConfigDict(extra="forbid") + traces: pydantic.AnyHttpUrl | None = None + logs: pydantic.AnyHttpUrl | None = None + metrics: pydantic.AnyHttpUrl | None = None + + _urls = pydantic.field_validator("traces", "logs", "metrics")(_credential_free_url) + + +class ObservabilityConfig(pydantic.BaseModel): + """Stable deployment identity and optional metadata-only diagnostic destinations.""" + + model_config = pydantic.ConfigDict(extra="forbid") + deployment_id: uuid.UUID + otlp_http_endpoints: TelemetryEndpoints | None = None + diagnostics_url: pydantic.AnyHttpUrl | None = None + + _url = pydantic.field_validator("diagnostics_url")(_credential_free_url) diff --git a/app/settings.py b/app/settings.py index 92e245e..8c392cf 100644 --- a/app/settings.py +++ b/app/settings.py @@ -25,6 +25,7 @@ class Settings(BaseSettings): env_file_encoding="utf-8", case_sensitive=False, extra="ignore", # Ignore extra environment variables + hide_input_in_errors=True, env_nested_delimiter="__", ) diff --git a/docker-compose.yml b/docker-compose.yml index b82609c..cd94936 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -56,7 +56,20 @@ services: DATABASE_URL: "postgresql+psycopg://inkcre_core:${CORE_DATABASE_PASSWORD:-local-core-database-password-at-least-32-bytes}@postgres:5432/${POSTGRES_DB:-inkcre}" INKCRE_ENV_FILE: "" JWT_SECRET: ${JWT_SECRET:-local-development-jwt-secret-at-least-32-bytes} - OBSRV__LOGGING_BACKEND: ${OBSRV__LOGGING_BACKEND:-none} + OBSRV__LOGGING_BACKEND: ${OBSRV__LOGGING_BACKEND:-postgresql} + OBSRV__TELEMETRY_ENABLED: ${OBSRV__TELEMETRY_ENABLED:-false} + OTEL_EXPORTER_OTLP_TRACES_ENDPOINT: ${OTEL_EXPORTER_OTLP_TRACES_ENDPOINT:-} + OTEL_EXPORTER_OTLP_LOGS_ENDPOINT: ${OTEL_EXPORTER_OTLP_LOGS_ENDPOINT:-} + OTEL_EXPORTER_OTLP_METRICS_ENDPOINT: ${OTEL_EXPORTER_OTLP_METRICS_ENDPOINT:-} + OTEL_EXPORTER_OTLP_HEADERS: ${OTEL_EXPORTER_OTLP_HEADERS:-} + OTEL_EXPORTER_OTLP_TIMEOUT: ${OTEL_EXPORTER_OTLP_TIMEOUT:-10} + OTEL_EXPORTER_OTLP_TRACES_TIMEOUT: ${OTEL_EXPORTER_OTLP_TRACES_TIMEOUT:-} + OTEL_EXPORTER_OTLP_LOGS_TIMEOUT: ${OTEL_EXPORTER_OTLP_LOGS_TIMEOUT:-} + OTEL_EXPORTER_OTLP_METRICS_TIMEOUT: ${OTEL_EXPORTER_OTLP_METRICS_TIMEOUT:-} + OTEL_PYTHON_SDK_INTERNAL_METRICS_ENABLED: ${OTEL_PYTHON_SDK_INTERNAL_METRICS_ENABLED:-true} + OTEL_EXPORTER_OTLP_TRACES_HEADERS: ${OTEL_EXPORTER_OTLP_TRACES_HEADERS:-} + OTEL_EXPORTER_OTLP_LOGS_HEADERS: ${OTEL_EXPORTER_OTLP_LOGS_HEADERS:-} + OTEL_EXPORTER_OTLP_METRICS_HEADERS: ${OTEL_EXPORTER_OTLP_METRICS_HEADERS:-} PORT: 8000 ports: - "127.0.0.1:${CORE_PORT:-8000}:8000" diff --git a/docs/40-deployment/README.md b/docs/40-deployment/README.md index cd78ccb..4eb8b36 100644 --- a/docs/40-deployment/README.md +++ b/docs/40-deployment/README.md @@ -8,6 +8,7 @@ GitHub workflow and composite-action YAML owns only GitHub event selection, perm - [development-environment.md](development-environment.md) - [agent-debug.md](agent-debug.md) +- [observability.md](observability.md) - [database-contract.md](database-contract.md) - [docker.md](docker.md) - [first-party-extension-distribution.md](first-party-extension-distribution.md) diff --git a/docs/40-deployment/observability.md b/docs/40-deployment/observability.md new file mode 100644 index 0000000..4a02a56 --- /dev/null +++ b/docs/40-deployment/observability.md @@ -0,0 +1,73 @@ +# 按需启用可观测性 + +新增遥测缺省关闭。现有 stdout、PostgreSQL 日志、`job.` 关联和 Job 日志查询继续工作;显式配置的 `none` 或 Logtail 也继续生效。开启遥测会增加标准 OTLP 元数据出口,不替换 PG writer,不转发历史日志或任意 Python logger。共享行为由[跨 Peer 可观测性契约](../_shared/20-product-tdd/observability-contract.md)拥有。 + +## 配置与启停 + +每个 Python Peer 在运行环境或本地 `.env` 设置 `OBSRV__TELEMETRY_ENABLED=true` 才启用。配置 endpoint、认证头或安装 SDK 都不能自动打开它。修改后重启进程;没有后台配置探测或热重载。关闭态不初始化新增 provider、导出线程或队列,也不查询新增共享配置。 + +部署 owner 在数据库已完成初始化后运行 `pdm run python scripts/observability.py`。该命令只在 `configs` 缺少 `inkcre.observability` 时插入一次 UUID;重复运行保留已存在的身份和配置,不能用于重置克隆后的身份。它不会启用任何 Peer。独立 preview、数据库克隆或新 owner 的部署,在启用前显式更换 deployment ID。 + +共享配置 schema 是 `inkcre.observability.v1`,包含 `deployment_id`,以及可选的 `otlp_http_endpoints`(`traces`、`logs`、`metrics` 完整 URL)和 `diagnostics_url`。共享 URL 禁止 userinfo、query 和 fragment。私密认证始终留在服务端运行配置,不写入共享 value、浏览器 bundle 或诊断链接。 + +```dotenv +OBSRV__TELEMETRY_ENABLED=true +OBSRV__LOGGING_BACKEND=postgresql +OTEL_EXPORTER_OTLP_TRACES_ENDPOINT=https://your-otlp-host/otlp/v1/traces +OTEL_EXPORTER_OTLP_LOGS_ENDPOINT=https://your-otlp-host/otlp/v1/logs +OTEL_EXPORTER_OTLP_METRICS_ENDPOINT=https://your-otlp-host/otlp/v1/metrics +OTEL_EXPORTER_OTLP_HEADERS="Authorization=Basic%20YOUR_ENCODED_VALUE" +``` + +当前出口为标准 OTLP/HTTP protobuf。三个 endpoint 使用完整地址,不自动添加路径;未配置的信号不导出。运行环境和 `.env` 的 per-signal endpoint 优先于共享默认值。认证支持公共 `OTEL_EXPORTER_OTLP_HEADERS` 和优先级更高的 `OTEL_EXPORTER_OTLP_{TRACES,LOGS,METRICS}_HEADERS`。服务名称、版本、运行实例由本实现提供,不读取任意 `OTEL_RESOURCE_ATTRIBUTES`,避免将配置原文引入 Resource。 + +Python 在数据库准入、Peer 注册之后读取一次共享配置,读取限时两秒;配置缺失、无效或读取失败时暂停新增遥测,不影响业务 readiness。单信号初始化失败只报告不含原始配置的本地诊断,不改变业务 readiness 或 PG 生命周期。配置恢复后需要重启,不自动轮询重试。 + +Grafana Cloud Free 是当前首个部署目标。`Authorization` 的 Basic 值必须是 `base64(instance ID:token)`,不能在 `Basic%20` 后直接拼原始 `glc_` token。Cloud Portal 的 stack 页面中,使用 OpenTelemetry 的 Configure 获取基础 endpoint 和认证变量;per-signal 地址应追加各自的 `/v1/traces`、`/v1/logs`、`/v1/metrics`,Python 的 Basic 空格使用 `%20`。[Grafana 官方说明](https://grafana.com/docs/grafana-cloud/send-data/otlp/send-data-otlp/)。账号准入、实际免费额度和查询读回必须在目标 stack 单独验收;配置示例不证明云端已通过。不要为本接入启用付费计划或依赖试用专属功能。 + +## 浏览器转发 + +浏览器不能使用上述私密 Grafana 写入凭据。client-web 的本地连接配置单独控制启用;需要服务端转发时,显式配置 relay 基础 URL,例如 `https://core.example/telemetry`,由客户端复用该连接的现有短效 Peer JWT。没有 relay 时,只能使用已准入、适合公开客户端的出口。relay URL 本身不启用遥测,也不改变业务的数据库直写或 Peer 路由。 + +core-py 的 `POST /telemetry/v1/{traces,logs,metrics}` 复用普通 Peer JWT 认证,仅向自身已成功启用的对应信号出口发送。调用方不能指定目标或转发认证头;服务端使用本地私密认证。它仅接受标准 OTLP/HTTP protobuf,直接使用标准 protobuf 请求与响应;JSON 返回 415。上游的 200 或 204 视为接收成功,204 的空响应映射为空 protobuf 成功响应;仍须独立查询验证持久化。client-web 使用官方 protobuf exporter,避免应用自行实现 OTLP JSON 的十六进制 Trace/Span ID 规则。不增加持久队列、后台重试或任意 URL 代理。 + +每个请求最多 256 KiB;每个进程最多四个转发请求同时进行,上游 HTTP 使用对应信号的 timeout;请求读取和发送的总预算为该 timeout 加两秒(最多三十二秒),不跟随重定向。不支持压缩请求。关闭或目的地不可用返回 503,满额返回 429,超大返回 413,格式错误返回 400,转发失败返回固定的 502,不回显上游错误体。转发会唤醒其宿主并消耗按请求资源,因此必须由部署显式选择;它不承诺遥测零额外运行费用。 + +## 首批采集与查询 + +| 边界 | 可观察内容 | +| --- | --- | +| `http.server` / `peer.http` | 路由模板、状态、耗时;Peer 目标及未执行、结果未知或已响应,不导出 URL query、headers 或 body | +| `job.submit` / `job.execute` | Job ID、受限提交 carrier、独立执行 Trace 与 Span Link;提交写入 span 只证明事务内 flush,`job.submitted` / `job.closed` 事件在本地拥有的事务提交后发出 | +| `cron.materialize` | 每次检查的 Cron ID 和当次实际创建的 Job,不复用最初创建 Cron 的请求上下文 | +| `ai.chat` / `ai.embed` | 共用 AI 执行边界;配置的模型标识、provider/dialect、耗时、真实 usage 和受控结束原因 | +| `agent.turn` / `agent.step` / `agent.tool` | Thread、turn、步骤、并发父子关系与工具结果;动态业务 ID 不进入指标维度 | +| Agent 检索、读取与结果 | 分开记录候选实体、实际读到的实体、Resolver 成功读取及最后选中的引用;只导出受控 ID 和计数 | + +`inkcre.operation.count` 和 `inkcre.operation.duration` 记录操作范围的 `success`、`error`、`cancelled`,结果独立于是否开启 Trace。它们不是全库 Job 终态统计;过期收敛等未进入执行 scope 的路径不能从这个 counter 推算。请求 5xx、已捕获的 Job 失败/取消、工具错误结果在业务边界显式分类。耗时直方图按秒显式分桶,从 5ms 到 300s,覆盖 HTTP 短调用与较长 AI/Job 操作。指标只使用固定操作和结果等有界标签,不包含 Job、Trace、Thread、Peer 实例或动态模型 ID。 + +资源使用 `service.name`、`service.version`、`service.instance.id`、已知的 `inkcre.deployment.id` 与 `inkcre.peer.id`。查询单个任务用 `inkcre.job.id`,从执行 Trace 的 Link 查提交 Trace;同步 Peer 调用使用标准 W3C context。只有配置的 Peer 传输注入上下文,外部模型请求不注入内部 Trace Context。关闭的提交方写 NULL,关闭的执行方不创建 Trace,但领取和关闭保留已有 carrier。缺失、采样和过期导致的空缺不是业务未执行的证据。 + +生成的 Job carrier 各最多 512 UTF-8 bytes;SDK 注入的可选 tracestate 超限时整体省略并记录固定原因事件。SQL 容量约束拒绝直接超限输入。正式 migration `3d9593b0c855` 增加两列;nullable 不解除 exact-head 准入,部署需协调更新 schema、Python 与客户端,刷新 PostgREST schema cache。关闭遥测无需 downgrade 或删除任何 Job、PG 日志或 carrier。 + +## AI 数据边界 + +GenAI 映射固定到 `semantic-conventions-genai` 修订 `e07f4ebacb08f56db8c4c882d117720333fbca04`,由 AI owner 的 `telemetry.py` 维护。OpenAI-compatible 和 Alibaba adapter 在真实 SDK 回包处采集;Alibaba 的空 choices 尾块仍读取 usage,重复累计总量只记最后一次。input/output 缺失分别标记 `unavailable`,真实非负 int64(包括零)标记 `provider`。缓存和推理 token 不重复加到 input/output 总量。 + +`inkcre.ai.token.usage` 只累计已知值;缺失、负值、布尔与越界值不添加零或负样本,缺失另记 `inkcre.ai.usage.missing`。当前没有价格配置或成本估算,未返回的成本保持未知;Token 总和及采样 Trace 都不能冒充 provider 账单。 + +基础模式不导出 prompt、答案、工具输入输出、query、provider 配置、任意异常消息或 stack。配置中的模型标识受字符集与长度限制,provider 任意回显 ID/model 不复制;ToolCall ID 使用 SHA-256 摘要关联。最终引用最多导出前 128 项中的有效 ID,同时记录总数和未导出数量;不截断原业务结果。 + +首批来源事件分别是 `inkcre.retrieval.candidates`、`inkcre.entity.read` 和 `inkcre.agent.query.result`。最后一个事件描述选入 Job 内存 state 的结果,不提前声称最终数据库 close 已成功。当前点位覆盖 Agent 的 retrieve/get_entities/Resolver Tool 与 Agent Query Sink;直接非 Agent 检索、图遍历和 Resolver 内部传递读取尚无完整来源事件覆盖。关联不能证明模型确实依据该实体或答案正确,也不提供长期内容快照。原有 `agent_debug` 与 PG/Logtail 内容行为保持,不能将新出口的元数据策略泛称为整个部署没有原文副本。 + +## 故障、关闭与迁移后端 + +需要队列丢弃及导出观测值时,在进程环境设置 `OTEL_PYTHON_SDK_INTERNAL_METRICS_ENABLED=true`;Compose 默认值为 true,允许显式覆盖。原生启动用 `export` 或 dotenv 启动器注入,单由应用 Pydantic 读取 `.env` 不会改变 SDK 读取的进程环境。此变量不自动启用遥测。配置 metrics 出口后,原生 SDK 内部指标只保留组件类型、错误类型与 HTTP 状态;不导出 endpoint、组件随机名称。未配置 metrics 或显式关闭内部指标时只有本地固定 warning,没有 SDK 丢弃计数。SDK 队列并发检查可使计数与真正损失存在误差,指标本身也可能丢失;它们用于诊断,不是精确损失账本,缺少样本不能推断零丢弃。 + +Span 和 log 各使用 SDK 的 256 条有界队列、最多 256 条一批、五秒周期;metric 为三十秒 push,不增加 scrape 或唤醒归零实例的采集轮询。HTTP 导出 timeout 采用标准 `OTEL_EXPORTER_OTLP_TIMEOUT`,缺省十秒;按信号的 `OTEL_EXPORTER_OTLP_{TRACES,LOGS,METRICS}_TIMEOUT` 优先,允许有限正数且不超过三十秒。真实 Cloud 首次 TLS 往返超过一秒,固定一秒会误丢弃可用出口的数据;故障实验使用显式一秒配置。源端 SDK warning/导出失败只输出固定本地诊断,不转发 SDK 错误体或凭据。 + +正常关闭先完成业务资源和 PG writer,再停止接受新记录,并发关闭 trace/log SDK,最后关闭 metrics 以采集前两者的最终导出与丢弃计数。SDK 1.45 的 trace/log worker 等待窗口为三十秒,metrics shutdown 等待参数为对应信号 timeout 加两秒(缺省十二秒);这些不是网络总时限或交付保证。不先执行一次会忽略 timeout 的 force_flush。平台提前杀进程、请求后冻结或断网时允许丢失尾部记录;使用该平台前需要验证实际终止宽限期,不能延长每个业务请求来掩盖此限制。 + +更换后端只调整标准出口和服务端认证,重新验证 ID、Links、AI unknown/zero 与聚合;Grafana 的查询、datasource、面板及诊断链接留在部署/消费侧。历史数据、查询和面板不承诺无成本迁移。故障回退是关闭本地遥测开关后重启,PG 无需恢复或迁移,业务 schema 与历史记录保持。 + +查询验收使用独立 Viewer service account,经 Grafana 的 datasource query/proxy API 读取 Tempo、Loki、Prometheus;`GRAFANA_URL` 与 `GRAFANA_SERVICE_ACCOUNT_TOKEN` 仅供部署工具使用,不进入应用采集。写入成功必须再核对存储后的 ID、Links、AI 来源/未知值与指标。若本机 Python 没有默认 CA 路径,在进程环境用标准 `OTEL_EXPORTER_OTLP_CERTIFICATE` 指向可信 CA 文件,不能关闭 TLS 校验。 diff --git a/docs/40-deployment/runtime-orchestration.md b/docs/40-deployment/runtime-orchestration.md index 22793db..217e541 100644 --- a/docs/40-deployment/runtime-orchestration.md +++ b/docs/40-deployment/runtime-orchestration.md @@ -124,6 +124,9 @@ Readiness 复用 `app/database_contract` 的同步 psycopg 检查,在 `asyncio 关闭时先停止接受新 backend 记录,最多等待五秒排空;超时取消写入并丢弃剩余记录,之后才释放 连接池。日志仍为 best-effort telemetry,不承诺进程崩溃后的交付。导入模块不启动 writer、不连接数据库。 +新增标准遥测在显式启用时,于数据库准入和 Peer 注册后初始化;PG 路径保持独立。 +元数据来源、OTLP 出口、有界关闭和浏览器认证转发见[可观测性配置](observability.md)。 + ### 8. Each Peer may own a scheduler APScheduler belongs to each web process. Cron row serialization, occurrence identity and the conditional pending-Job diff --git a/docs/openapi.json b/docs/openapi.json index b8c1b20..9780169 100644 --- a/docs/openapi.json +++ b/docs/openapi.json @@ -4252,6 +4252,47 @@ } } } + }, + "/telemetry/v1/{signal}": { + "post": { + "tags": [ + "telemetry" + ], + "summary": "Export Telemetry", + "description": "Forward standard OTLP using this Peer's enabled private destination.\n\nThe core router requires the existing Peer JWT. The caller cannot choose an\nupstream URL or forward headers. No application payload or durable queue is created.", + "operationId": "export_telemetry_telemetry_v1__signal__post", + "parameters": [ + { + "name": "signal", + "in": "path", + "required": true, + "schema": { + "enum": [ + "traces", + "logs", + "metrics" + ], + "type": "string", + "title": "Signal" + } + } + ], + "responses": { + "200": { + "description": "Successful Response" + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + } + } } }, "components": { @@ -5403,6 +5444,28 @@ "type": "object", "title": "Parameters" }, + "submission_traceparent": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Submission Traceparent" + }, + "submission_tracestate": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Submission Tracestate" + }, "state": { "additionalProperties": true, "type": "object", diff --git a/libs/obsrv/main.py b/libs/obsrv/main.py index e96eb52..19e16ca 100644 --- a/libs/obsrv/main.py +++ b/libs/obsrv/main.py @@ -76,6 +76,9 @@ def start_obsrv() -> None: async def close_obsrv() -> None: from .log_handler_postgresql import PostgreSQLHandler + from .telemetry import close_telemetry + + await close_telemetry() for handler in LOGGER.handlers: if isinstance(handler, PostgreSQLHandler): diff --git a/libs/obsrv/setting.py b/libs/obsrv/setting.py index e9ab006..8e3a334 100644 --- a/libs/obsrv/setting.py +++ b/libs/obsrv/setting.py @@ -9,6 +9,11 @@ class ObsrvSetting(BaseSettings): """Observability settings.""" + telemetry_enabled: bool = Field( + default=False, + description="Enable optional metadata-only OpenTelemetry export for this Peer.", + ) + agent_debug: bool = Field( default=False, description="Record Agent inputs, tool contracts and execution events for development.", diff --git a/libs/obsrv/telemetry.py b/libs/obsrv/telemetry.py new file mode 100644 index 0000000..6343e9d --- /dev/null +++ b/libs/obsrv/telemetry.py @@ -0,0 +1,488 @@ +"""Opt-in OTLP metadata transport; callers own allowed field names and value sources.""" + +import asyncio +from collections.abc import Callable, Generator, Mapping, Sequence +from contextlib import ExitStack, contextmanager +from dataclasses import dataclass, field +from functools import partial +import logging +import math +import os +import sys +import time +from typing import TYPE_CHECKING, Literal +from urllib.parse import urlsplit +import uuid + +from opentelemetry.context import Context +from opentelemetry.trace import INVALID_SPAN, Link, Span, SpanKind, StatusCode, use_span +from opentelemetry.util.types import AttributeValue +from pydantic import SecretStr +from pydantic_settings import BaseSettings, SettingsConfigDict + +if TYPE_CHECKING: + from opentelemetry.sdk._logs import LoggerProvider + from opentelemetry.sdk.metrics import MeterProvider + from opentelemetry.sdk.trace import TracerProvider + from opentelemetry.metrics import Counter, Histogram + from opentelemetry.trace import Tracer + from opentelemetry._logs import Logger + + +class _ExportSettings(BaseSettings): + model_config = SettingsConfigDict(extra="ignore") + + otel_exporter_otlp_traces_endpoint: str = "" + otel_exporter_otlp_logs_endpoint: str = "" + otel_exporter_otlp_metrics_endpoint: str = "" + otel_exporter_otlp_headers: SecretStr = SecretStr("") + otel_exporter_otlp_traces_headers: SecretStr = SecretStr("") + otel_exporter_otlp_logs_headers: SecretStr = SecretStr("") + otel_exporter_otlp_metrics_headers: SecretStr = SecretStr("") + # Validate each signal separately, so a bad timeout does not disable siblings. + otel_exporter_otlp_timeout: str = "10" + otel_exporter_otlp_traces_timeout: str = "" + otel_exporter_otlp_logs_timeout: str = "" + otel_exporter_otlp_metrics_timeout: str = "" + + +@dataclass +class _Runtime: + connections: dict[str, tuple[str, dict[str, str], float]] = field(default_factory=dict) + traces: "TracerProvider | None" = None + logs: "LoggerProvider | None" = None + metrics: "MeterProvider | None" = None + tracer: "Tracer | None" = None + logger: "Logger | None" = None + operations: "Counter | None" = None + duration: "Histogram | None" = None + tokens: "Counter | None" = None + missing_usage: "Counter | None" = None + + +_runtime: _Runtime | None = None + + +def _diagnostic(reason: str) -> None: + # This sink never receives endpoints, headers, exceptions, or arbitrary log messages. + try: + print(f"OpenTelemetry: {reason}", file=sys.stderr) + except Exception: + # Diagnostics cannot make a broken stderr stream a business failure. + pass + + +class _SDKDiagnostics(logging.Handler): + def emit(self, record: logging.LogRecord) -> None: + # SDK errors may contain response bodies, URLs or malformed secret headers. + _diagnostic("SDK warning or export failure; check the configured OTLP receiver") + + +def _configure_diagnostics() -> None: + logger = logging.getLogger("opentelemetry") + if not any(isinstance(handler, _SDKDiagnostics) for handler in logger.handlers): + handler = _SDKDiagnostics(level=logging.WARNING) + logger.addHandler(handler) + logger.propagate = False + + +def _endpoint(value: str) -> str: + parsed = urlsplit(value) + if ( + parsed.scheme not in ("http", "https") + or not parsed.hostname + or not parsed.path + or parsed.username is not None + or parsed.password is not None + or parsed.query + or parsed.fragment + or any(character.isspace() for character in value) + ): + raise ValueError("invalid OTLP URL") + # Accessing port performs urllib's range and syntax validation. + parsed.port + return value + + +def is_enabled() -> bool: + """Whether at least one metadata signal initialized for this runtime.""" + return _runtime is not None + + +def get_export_connection(signal: str) -> tuple[str, dict[str, str], float] | None: + """Return private endpoint, copied headers and timeout seconds for server forwarding.""" + connection = _runtime.connections.get(signal) if _runtime is not None else None + return ( + (connection[0], dict(connection[1]), connection[2]) if connection is not None else None + ) + + +def start_telemetry( + *, + enabled: bool, + service_version: str, + resource_attributes: Mapping[str, AttributeValue] | None = None, + endpoints: Mapping[str, str] | None = None, +) -> None: + """Initialize only configured signals; configuration failures never gate readiness.""" + global _runtime + if not enabled or _runtime is not None: + return + + _configure_diagnostics() + try: + config = _ExportSettings(_env_file=os.getenv("INKCRE_ENV_FILE", ".env") or None) + except Exception: + _diagnostic("invalid runtime export configuration; telemetry is disabled") + return + + from opentelemetry.exporter.otlp.proto.http import Compression + from opentelemetry.exporter.otlp.proto.http._log_exporter import OTLPLogExporter + from opentelemetry.exporter.otlp.proto.http.metric_exporter import OTLPMetricExporter + from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter + from opentelemetry.metrics import NoOpMeterProvider + from opentelemetry.sdk._logs import LoggerProvider, LogRecordLimits + from opentelemetry.sdk._logs.export import BatchLogRecordProcessor + from opentelemetry.sdk.metrics import MeterProvider, SimpleFixedSizeExemplarReservoir + from opentelemetry.sdk.metrics.export import PeriodicExportingMetricReader + from opentelemetry.sdk.metrics.view import ExplicitBucketHistogramAggregation, View + from opentelemetry.sdk.resources import Resource + from opentelemetry.sdk.trace import SpanLimits, TracerProvider + from opentelemetry.sdk.trace.export import BatchSpanProcessor + from opentelemetry.sdk.trace.sampling import ALWAYS_ON + from opentelemetry.util.re import parse_env_headers + from requests.utils import check_header_validity # type: ignore[untyped-import] + + # Resource.create invokes environment detectors; direct construction is deliberate. + resource = Resource( + { + **(resource_attributes or {}), + "service.name": "core-py", + "service.version": service_version, + "service.instance.id": str(uuid.uuid4()), + } + ) + runtime = _Runtime() + no_metrics = NoOpMeterProvider() + for signal in ("metrics", "traces", "logs"): + endpoint = getattr(config, f"otel_exporter_otlp_{signal}_endpoint") or ( + endpoints or {} + ).get(signal, "") + if not endpoint: + continue + cleanup: Callable[[], None] | None = None + internal_metrics = runtime.metrics or no_metrics + try: + endpoint = _endpoint(endpoint) + timeout = float( + getattr(config, f"otel_exporter_otlp_{signal}_timeout") + or config.otel_exporter_otlp_timeout + or "10" + ) + if not math.isfinite(timeout) or not 0 < timeout <= 30: + raise ValueError("OTLP timeout must be finite, positive and at most 30 seconds") + raw_headers = ( + getattr(config, f"otel_exporter_otlp_{signal}_headers").get_secret_value() + or config.otel_exporter_otlp_headers.get_secret_value() + ) + headers = parse_env_headers(raw_headers, liberal=True) + # The SDK parser drops malformed entries. Do not silently accept partial auth. + if len(headers) != len([part for part in raw_headers.split(",") if part.strip()]): + raise ValueError("invalid OTLP headers") + for header in headers.items(): + check_header_validity(header) + + if signal == "traces": + exporter = OTLPSpanExporter( + endpoint=endpoint, + headers=headers, + timeout=timeout, + compression=Compression.NoCompression, + meter_provider=internal_metrics, + ) + cleanup = exporter.shutdown + traces = TracerProvider( + resource=resource, + sampler=ALWAYS_ON, + shutdown_on_exit=False, + span_limits=SpanLimits( + max_attributes=64, + max_events=16, + max_links=16, + max_attribute_length=256, + max_span_attributes=64, + max_event_attributes=64, + max_link_attributes=64, + max_span_attribute_length=256, + ), + meter_provider=internal_metrics, + ) + processor = BatchSpanProcessor( + exporter, + max_queue_size=256, + max_export_batch_size=256, + schedule_delay_millis=5000, + export_timeout_millis=timeout * 1000, + meter_provider=internal_metrics, + ) + cleanup = processor.shutdown + traces.add_span_processor(processor) + cleanup = traces.shutdown + tracer = traces.get_tracer("inkcre.metadata") + runtime.traces, runtime.tracer = traces, tracer + elif signal == "logs": + log_exporter = OTLPLogExporter( + endpoint=endpoint, + headers=headers, + timeout=timeout, + compression=Compression.NoCompression, + meter_provider=internal_metrics, + ) + cleanup = log_exporter.shutdown + logs = LoggerProvider( + resource=resource, + shutdown_on_exit=False, + meter_provider=internal_metrics, + log_record_limits=LogRecordLimits(max_attributes=64, max_attribute_length=256), + ) + log_processor = BatchLogRecordProcessor( + log_exporter, + max_queue_size=256, + max_export_batch_size=256, + schedule_delay_millis=5000, + export_timeout_millis=timeout * 1000, + meter_provider=internal_metrics, + ) + cleanup = log_processor.shutdown + logs.add_log_record_processor(log_processor) + cleanup = logs.shutdown + logger = logs.get_logger("inkcre.metadata") + runtime.logs, runtime.logger = logs, logger + else: + metric_exporter = OTLPMetricExporter( + endpoint=endpoint, + headers=headers, + timeout=timeout, + compression=Compression.NoCompression, + meter_provider=no_metrics, + ) + cleanup = metric_exporter.shutdown + reader = PeriodicExportingMetricReader( + metric_exporter, + export_interval_millis=30000, + export_timeout_millis=timeout * 1000, + ) + cleanup = partial(reader.shutdown, timeout_millis=(timeout + 2) * 1000) + metrics = MeterProvider( + resource=resource, + metric_readers=[reader], + shutdown_on_exit=False, + views=[ + # SDK defaults are too coarse for a duration measured in seconds. + View( + instrument_name="inkcre.operation.duration", + aggregation=ExplicitBucketHistogramAggregation( + boundaries=( + 0.005, + 0.01, + 0.025, + 0.05, + 0.1, + 0.25, + 0.5, + 1, + 2.5, + 5, + 10, + 30, + 60, + 300, + ) + ), + ), + View( + meter_name="opentelemetry-sdk", + attribute_keys={ + "otel.component.type", + "error.type", + "http.response.status_code", + }, + # An exemplar can otherwise retain attributes excluded by the View. + exemplar_reservoir_factory=lambda _: ( + lambda _config=None, **_kwargs: SimpleFixedSizeExemplarReservoir(size=0) + ), + ), + ], + ) + cleanup = partial(metrics.shutdown, timeout_millis=(timeout + 2) * 1000) + meter = metrics.get_meter("inkcre.metadata") + operations = meter.create_counter("inkcre.operation.count", unit="{operation}") + duration = meter.create_histogram("inkcre.operation.duration", unit="s") + tokens = meter.create_counter("inkcre.ai.token.usage", unit="{token}") + missing_usage = meter.create_counter("inkcre.ai.usage.missing", unit="{field}") + runtime.metrics = metrics + runtime.operations, runtime.duration = operations, duration + runtime.tokens, runtime.missing_usage = tokens, missing_usage + except Exception: + if cleanup is not None: + _shutdown(cleanup) + _diagnostic(f"{signal} configuration or initialization failed; signal is disabled") + else: + runtime.connections[signal] = (endpoint, dict(headers), timeout) + + if any((runtime.traces, runtime.logs, runtime.metrics)): + _runtime = runtime + else: + _diagnostic("no usable per-signal endpoint; telemetry is disabled") + + +def _shutdown(shutdown: Callable[[], None]) -> None: + try: + shutdown() + except Exception: + _diagnostic("signal shutdown failed; remaining telemetry may be lost") + + +async def close_telemetry() -> None: + """Stop new records and concurrently drain SDK queues outside the event loop. + + SDK trace/log shutdown waits up to 30 seconds for each worker, then metrics gets + its configured request timeout plus two seconds. These worker wait budgets are + not transport-wide hard deadlines; freezing or termination can lose records. + """ + global _runtime + runtime, _runtime = _runtime, None + if runtime is None: + return + shutdowns: list[Callable[[], None]] = [] + for provider in (runtime.traces, runtime.logs): + if provider is not None: + shutdowns.append(provider.shutdown) + # SDK 1.45 force_flush ignores its timeout; shutdown owns the bounded queue drain. + await asyncio.gather(*(asyncio.to_thread(_shutdown, shutdown) for shutdown in shutdowns)) + if runtime.metrics is not None: + # Collect final native processor/exporter metrics after their workers have drained. + await asyncio.to_thread( + _shutdown, + partial( + runtime.metrics.shutdown, + timeout_millis=(runtime.connections["metrics"][2] + 2) * 1000, + ), + ) + + +def _record(action: Callable[[], object]) -> None: + """Isolate one owned telemetry call; never wrap the caller's business block.""" + try: + action() + except Exception: + _diagnostic("recording failed; optional telemetry may be lost") + + +@dataclass +class Operation: + """One execution scope's result, independent of whether its Span is recorded. + + Business boundaries set outcome when they return a failure or handle cancellation. + Span methods remain available explicitly through span; Span status is not a metric. + Those standard SDK methods require the caller's admitted primitive metadata values; + the facade isolates its own SDK calls, not arbitrary caller code inside the scope. + """ + + span: Span + outcome: Literal["success", "error", "cancelled"] = "success" + + +@contextmanager +def operation( + name: str, + *, + attributes: Mapping[str, AttributeValue] | None = None, + context: Context | None = None, + links: Sequence[Link] | None = None, + kind: SpanKind = SpanKind.INTERNAL, +) -> Generator[Operation, None, None]: + """Measure a fixed operation name; callers supply only admitted metadata values. + + Exceptions and cancellation propagate unchanged. Their messages, stack traces and + status descriptions are never recorded by this context manager. + """ + runtime = _runtime + if runtime is None: + yield Operation(INVALID_SPAN) + return + + span = INVALID_SPAN + if runtime.tracer is not None: + try: + span = runtime.tracer.start_span( + name, + attributes=attributes, + context=context, + links=links, + kind=kind, + record_exception=False, + set_status_on_exception=False, + ) + except Exception: + _diagnostic("span creation failed; operation metrics remain available") + observation = Operation(span) + started = time.perf_counter() + activation = ExitStack() + _record( + lambda: activation.enter_context( + use_span(span, record_exception=False, set_status_on_exception=False) + ) + ) + try: + yield observation + except BaseException as error: + observation.outcome = ( + "cancelled" if isinstance(error, asyncio.CancelledError) else "error" + ) + _record(partial(span.set_attribute, "error.type", type(error).__name__)) + raise + finally: + # Exit the SDK context with no exception and end separately. This prevents SDK + # context cleanup or span processors from masking the caller's original error. + if observation.outcome != "success": + _record(partial(span.set_status, StatusCode.ERROR)) + _record(partial(span.set_attribute, "inkcre.outcome", observation.outcome)) + labels = {"operation": name, "outcome": observation.outcome} + if runtime.operations is not None: + _record(partial(runtime.operations.add, 1, labels)) + if runtime.duration is not None: + _record(partial(runtime.duration.record, time.perf_counter() - started, labels)) + _record(activation.close) + _record(span.end) + + +def emit_event(name: str, attributes: Mapping[str, AttributeValue]) -> None: + """Emit a fixed event name and admitted metadata without a Python logging bridge.""" + runtime = _runtime + if runtime is not None and runtime.logger is not None: + from opentelemetry._logs import SeverityNumber + + _record( + partial( + runtime.logger.emit, + body=name, + severity_number=SeverityNumber.INFO, + attributes=attributes, + ) + ) + + +def record_ai_usage( + operation: str, input_tokens: int | None, output_tokens: int | None +) -> None: + """Count known nonnegative provider usage and count missing fields separately.""" + runtime = _runtime + if runtime is None or runtime.tokens is None or runtime.missing_usage is None: + return + for kind, value in (("input", input_tokens), ("output", output_tokens)): + labels = {"operation": operation, "token.type": kind} + if isinstance(value, int) and not isinstance(value, bool) and value >= 0: + _record(partial(runtime.tokens.add, value, labels)) + else: + _record(partial(runtime.missing_usage.add, 1, labels)) diff --git a/migrations/revision-integrity.json b/migrations/revision-integrity.json index aff994e..fe61da3 100644 --- a/migrations/revision-integrity.json +++ b/migrations/revision-integrity.json @@ -5,6 +5,7 @@ "revisions": { "143c4f4adc85_add_sink_catalog_and_instances.py": "dc5151df7c73ec84077f7698cfb1e7fdd5596978270055de2812da539d02043d", "1e4c7a9b2d5f_add_block_lexical_records.py": "20cd7c361b0f822811d52beb692a255078e3842fd64d74287ee0c13d0bc8df26", + "3d9593b0c855_add_job_submission_trace_context.py": "60d61b798da850ea24d301db3f97cb07e0d8d9695a6c5a9763b9ef4f990d0845", "3f7a9c2d5e1b_merge_extension_registry_feature_retrieval.py": "a12ebb2f37a00dcc13b6dea6b7f9be0f6870d57530e8d234eb934c080d027c07", "50b2c08dd267_add_relation_endpoint_indexes.py": "56a0dbfd5e032fa57b125e478940ff89254c167b38fe0a7a98046498577d13e5", "77cd53ad8080_add_global_jobs_crons_and_source_runtime.py": "8eda54520a7099c957ee6a1b5e6e48d5ee2b3dbe18b0dea4b6a5246dbef04bd2", diff --git a/migrations/versions/3d9593b0c855_add_job_submission_trace_context.py b/migrations/versions/3d9593b0c855_add_job_submission_trace_context.py new file mode 100644 index 0000000..ab1be50 --- /dev/null +++ b/migrations/versions/3d9593b0c855_add_job_submission_trace_context.py @@ -0,0 +1,57 @@ +"""add_job_submission_trace_context + +Revision ID: 3d9593b0c855 +Revises: a0465e3b028f +Create Date: 2026-10-03 17:23:07.957043 + +""" + +from typing import Sequence + +from alembic import op +import sqlalchemy as sa + + +# revision identifiers, used by Alembic. +revision: str = "3d9593b0c855" +down_revision: str | Sequence[str] | None = "a0465e3b028f" +branch_labels: str | Sequence[str] | None = None +depends_on: str | Sequence[str] | None = None + + +def upgrade() -> None: + """Upgrade schema.""" + # ### commands auto generated by Alembic - please adjust! ### + op.add_column( + "jobs", sa.Column("submission_traceparent", sa.Text(), nullable=True), schema="inkcre" + ) + op.add_column( + "jobs", sa.Column("submission_tracestate", sa.Text(), nullable=True), schema="inkcre" + ) + op.create_check_constraint( + "jobs_submission_traceparent_capacity", + "jobs", + "octet_length(submission_traceparent) <= 512", + schema="inkcre", + ) + op.create_check_constraint( + "jobs_submission_tracestate_capacity", + "jobs", + "octet_length(submission_tracestate) <= 512", + schema="inkcre", + ) + # ### end Alembic commands ### + + +def downgrade() -> None: + """Downgrade schema.""" + # ### commands auto generated by Alembic - please adjust! ### + op.drop_constraint( + "jobs_submission_tracestate_capacity", "jobs", schema="inkcre", type_="check" + ) + op.drop_constraint( + "jobs_submission_traceparent_capacity", "jobs", schema="inkcre", type_="check" + ) + op.drop_column("jobs", "submission_tracestate", schema="inkcre") + op.drop_column("jobs", "submission_traceparent", schema="inkcre") + # ### end Alembic commands ### diff --git a/pdm.lock b/pdm.lock index 3d4e6fc..1ef3339 100644 --- a/pdm.lock +++ b/pdm.lock @@ -5,7 +5,7 @@ groups = ["default", "dev", "extension-preview", "extension-publisher"] strategy = ["inherit_metadata"] lock_version = "4.5.0" -content_hash = "sha256:1c5c63a79fb7fd28d9fa2337a1a7ef1c0690c2f81dcf1ac360b667ce1dec0ee0" +content_hash = "sha256:a35a2bf8c4d178b0ddeea60b4c4256e1817073eb6c137e80f8c1bfbf1c1154a4" [[metadata.targets]] requires_python = ">=3.12,<3.13" @@ -630,6 +630,20 @@ files = [ {file = "frozenlist-1.8.0.tar.gz", hash = "sha256:3ede829ed8d842f6cd48fc7081d7a41001a56f1f38603f9d49bf3020d59a31ad"}, ] +[[package]] +name = "googleapis-common-protos" +version = "1.75.5" +requires_python = ">=3.10" +summary = "Common protobufs used in Google APIs" +groups = ["default"] +dependencies = [ + "protobuf<8.0.0,>=6.33.5", +] +files = [ + {file = "googleapis_common_protos-1.75.5-py3-none-any.whl", hash = "sha256:d7285525c23039db98f2463e6d5a4f9b958b94d497f03a844ece3259c4e72d5d"}, + {file = "googleapis_common_protos-1.75.5.tar.gz", hash = "sha256:c7a866fc34ed29a3b10af627a4b9b1dc2433313ca6e959f0ae4feb132047ed72"}, +] + [[package]] name = "greenlet" version = "3.5.5" @@ -1308,7 +1322,7 @@ files = [ [[package]] name = "opentelemetry-api" -version = "1.44.0" +version = "1.45.0" requires_python = ">=3.10" summary = "OpenTelemetry Python API" groups = ["default"] @@ -1316,8 +1330,132 @@ dependencies = [ "typing-extensions>=4.5.0", ] files = [ - {file = "opentelemetry_api-1.44.0-py3-none-any.whl", hash = "sha256:94b98c893a91b88657eaac1e3ba89618cdb85be6918196705354f34728b2cdef"}, - {file = "opentelemetry_api-1.44.0.tar.gz", hash = "sha256:67647e5e9566edcf421166fdf022b3537f818635daa852b289e34604dc6fb33a"}, + {file = "opentelemetry_api-1.45.0-py3-none-any.whl", hash = "sha256:80e068aba7cd56c8b58512d6a36f8d25cb1dfaa0c0a4cc1c938ccf9f362d9cb3"}, + {file = "opentelemetry_api-1.45.0.tar.gz", hash = "sha256:711ede81773c8025c2c03dac0450bc89f3d30aea6eabcc815c570d4e35a963f7"}, +] + +[[package]] +name = "opentelemetry-exporter-http-transport" +version = "0.66b0" +requires_python = ">=3.10" +summary = "OpenTelemetry Exporters HTTP transport" +groups = ["default"] +dependencies = [ + "opentelemetry-api~=1.15", +] +files = [ + {file = "opentelemetry_exporter_http_transport-0.66b0-py3-none-any.whl", hash = "sha256:c82689928a11505a2a0a26a04afe0ce06d270d3db984bc16c1a80ffa902db00e"}, + {file = "opentelemetry_exporter_http_transport-0.66b0.tar.gz", hash = "sha256:2c229b6593eaa22c86d9b8a15843dc23b406dbda00fb138339189aab07923b4e"}, +] + +[[package]] +name = "opentelemetry-exporter-http-transport" +version = "0.66b0" +extras = ["urllib3"] +requires_python = ">=3.10" +summary = "OpenTelemetry Exporters HTTP transport" +groups = ["default"] +dependencies = [ + "opentelemetry-exporter-http-transport==0.66b0", + "urllib3>=1.26", +] +files = [ + {file = "opentelemetry_exporter_http_transport-0.66b0-py3-none-any.whl", hash = "sha256:c82689928a11505a2a0a26a04afe0ce06d270d3db984bc16c1a80ffa902db00e"}, + {file = "opentelemetry_exporter_http_transport-0.66b0.tar.gz", hash = "sha256:2c229b6593eaa22c86d9b8a15843dc23b406dbda00fb138339189aab07923b4e"}, +] + +[[package]] +name = "opentelemetry-exporter-otlp-common" +version = "0.66b0" +requires_python = ">=3.10" +summary = "OpenTelemetry OTLP HTTP export utilities" +groups = ["default"] +dependencies = [ + "opentelemetry-sdk~=1.45.0", +] +files = [ + {file = "opentelemetry_exporter_otlp_common-0.66b0-py3-none-any.whl", hash = "sha256:35d24c867310f4713a9738b0202b6bade77955418fccb9e2f7eea01b76263916"}, + {file = "opentelemetry_exporter_otlp_common-0.66b0.tar.gz", hash = "sha256:362268ec6aa705e183776ff938539df1e8ce45bc5509d242538b1d40c26fe6a6"}, +] + +[[package]] +name = "opentelemetry-exporter-otlp-proto-common" +version = "1.45.0" +requires_python = ">=3.10" +summary = "OpenTelemetry Protobuf encoding" +groups = ["default"] +dependencies = [ + "opentelemetry-proto==1.45.0", +] +files = [ + {file = "opentelemetry_exporter_otlp_proto_common-1.45.0-py3-none-any.whl", hash = "sha256:7e1410ae6d3ed301f7a74bec96467d519d3f37e52f9d95b309ae125e5a1af863"}, + {file = "opentelemetry_exporter_otlp_proto_common-1.45.0.tar.gz", hash = "sha256:36495115a0c6a7aa946cfda9d59b6ed4e917b6ab0f75cdaf66bc1b176ec1be1f"}, +] + +[[package]] +name = "opentelemetry-exporter-otlp-proto-http" +version = "1.45.0" +requires_python = ">=3.10" +summary = "OpenTelemetry Collector Protobuf over HTTP Exporter" +groups = ["default"] +dependencies = [ + "googleapis-common-protos~=1.52", + "opentelemetry-api~=1.15", + "opentelemetry-exporter-http-transport[urllib3]==0.66b0", + "opentelemetry-exporter-otlp-common==0.66b0", + "opentelemetry-exporter-otlp-proto-common==1.45.0", + "opentelemetry-proto==1.45.0", + "opentelemetry-sdk~=1.45.0", + "typing-extensions>=4.5.0", +] +files = [ + {file = "opentelemetry_exporter_otlp_proto_http-1.45.0-py3-none-any.whl", hash = "sha256:74385f99266dafdefff41f3da4b76f6dba79d237a9d4487e6ff1ee653c74a0a3"}, + {file = "opentelemetry_exporter_otlp_proto_http-1.45.0.tar.gz", hash = "sha256:2f35496d96809f946f41b8805e6b93aec6c9b71b5fd759b75b6af4c084d992ae"}, +] + +[[package]] +name = "opentelemetry-proto" +version = "1.45.0" +requires_python = ">=3.10" +summary = "OpenTelemetry Python Proto" +groups = ["default"] +dependencies = [ + "protobuf<8.0,>=5.0", +] +files = [ + {file = "opentelemetry_proto-1.45.0-py3-none-any.whl", hash = "sha256:9731566359d7b8e1ee1e149d8e4ad6865bab965e657ad3387ec2784d16927666"}, + {file = "opentelemetry_proto-1.45.0.tar.gz", hash = "sha256:96ee414f24bc3f61ea8e17dc56b4348d4049d73db3eb17c6b3edf75b5b403300"}, +] + +[[package]] +name = "opentelemetry-sdk" +version = "1.45.0" +requires_python = ">=3.10" +summary = "OpenTelemetry Python SDK" +groups = ["default"] +dependencies = [ + "opentelemetry-api==1.45.0", + "opentelemetry-semantic-conventions==0.66b0", + "typing-extensions>=4.5.0", +] +files = [ + {file = "opentelemetry_sdk-1.45.0-py3-none-any.whl", hash = "sha256:5dc634c946546f61b757c5b1781f9fd1a10e96ffaf7357c797e508fe9c57e75e"}, + {file = "opentelemetry_sdk-1.45.0.tar.gz", hash = "sha256:20caa5130505e386c67c3da1c76e446c842698ced54c76c6148679539aa97972"}, +] + +[[package]] +name = "opentelemetry-semantic-conventions" +version = "0.66b0" +requires_python = ">=3.10" +summary = "OpenTelemetry Semantic Conventions" +groups = ["default"] +dependencies = [ + "opentelemetry-api==1.45.0", + "typing-extensions>=4.5.0", +] +files = [ + {file = "opentelemetry_semantic_conventions-0.66b0-py3-none-any.whl", hash = "sha256:175b19dd98c4473f4f43a2b1df59186fd7b4a48cd3f77cbe03438b6d6fda230a"}, + {file = "opentelemetry_semantic_conventions-0.66b0.tar.gz", hash = "sha256:97a77dce484c54861e7eeff7651fd8a806dd3c30e501dc316730215ec36890e6"}, ] [[package]] @@ -1453,6 +1591,23 @@ files = [ {file = "propcache-0.5.2.tar.gz", hash = "sha256:01c4fc7480cd0598bb4b57022df55b9ca296da7fc5a8760bd8451a7e63a7d427"}, ] +[[package]] +name = "protobuf" +version = "7.36.2" +requires_python = ">=3.10" +summary = "" +groups = ["default"] +files = [ + {file = "protobuf-7.36.2-cp310-abi3-macosx_10_9_universal2.whl", hash = "sha256:cbc70b17ee27e28894c7fee8bb04be1abead49e936bc70eb60052531eee2079e"}, + {file = "protobuf-7.36.2-cp310-abi3-manylinux2014_aarch64.whl", hash = "sha256:e11e1f0180583a2af89db6a2ecd9e8dc40aa6d2988ca175bfd0e6d12ea72d74e"}, + {file = "protobuf-7.36.2-cp310-abi3-manylinux2014_s390x.whl", hash = "sha256:f4fee11ec330d238b34a05c9b675f693c20415d1c5bd7d5320cc2f8a798eb9cf"}, + {file = "protobuf-7.36.2-cp310-abi3-manylinux2014_x86_64.whl", hash = "sha256:89f23aa53c24553a2416fd4fd1ec06f74fa42b14b546d8883128813f775bbfd2"}, + {file = "protobuf-7.36.2-cp310-abi3-win32.whl", hash = "sha256:912c1221170e16c08d1f086762f563dd61ff83c18b5fa6652952dfaded66f728"}, + {file = "protobuf-7.36.2-cp310-abi3-win_amd64.whl", hash = "sha256:a300819d441e078a5608c0d3c709796bb548136058fda017ae51d425b44fd353"}, + {file = "protobuf-7.36.2-py3-none-any.whl", hash = "sha256:bdb3a345d48db958e6ce1f18e508beb0cc981d64f24088427549c866cd039f1e"}, + {file = "protobuf-7.36.2.tar.gz", hash = "sha256:497d0463ff3316681da6c0b9e8d06cb465d61abce00b613ab42226175644d1bb"}, +] + [[package]] name = "psycopg" version = "3.3.4" diff --git a/pyproject.toml b/pyproject.toml index 165b26d..97b16f3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -51,6 +51,8 @@ dependencies = [ "croniter<7,>=6", "imapclient<4,>=3.1", "mcp>=2.1.1,<2.2.0", + "opentelemetry-sdk==1.45.0", + "opentelemetry-exporter-otlp-proto-http==1.45.0", ] [dependency-groups] diff --git a/run.py b/run.py index 22f2b74..bcc35e3 100644 --- a/run.py +++ b/run.py @@ -50,6 +50,7 @@ from app.routes.semantic_retrieval import PEER_INBOUND as semantic_retrieval_peer_inbound from app.routes.semantic_retrieval import ROUTER as semantic_retrieval_router from app.routes.sink import ROUTER as sink_router +from app.routes.telemetry import ROUTER as telemetry_router from app.business.source import SourceManager from app.business.cron import CronManager from app.business.job import JobManager @@ -72,6 +73,7 @@ SynthesisJobHandler, ) from app.middleware import LoggingMiddleware, require_peer_jwt +from app.observability import TelemetryMiddleware, initialize_telemetry from app.schemas.peer import PEER_EXECUTION_HEADER from app.health import check_database_readiness from app.runtime import RUNTIME_STATUS, RuntimePhase @@ -112,6 +114,8 @@ async def bootstrap_runtime(app: fastapi.FastAPI) -> None: PeerManager.register_inbound(organization_peer_inbound) PeerManager.register_inbound(extension_peer_inbound) + await initialize_telemetry() + # Core decoders exist independently of installed/enabled extensions. with bootstrap_step("core_registration"): register_core_resolvers() @@ -248,6 +252,7 @@ async def lifespan(app: fastapi.FastAPI): # 添加日志中间件 api_app.add_middleware(LoggingMiddleware) +api_app.add_middleware(TelemetryMiddleware) # 添加CORS中间件以支持跨域请求 api_app.add_middleware( @@ -309,6 +314,7 @@ async def readiness() -> JSONResponse: core_router.include_router(semantic_retrieval_router) core_router.include_router(lexical_retrieval_router) core_router.include_router(sink_router) +core_router.include_router(telemetry_router) api_app.include_router(core_router) if __name__ == "__main__": diff --git a/scripts/observability.py b/scripts/observability.py new file mode 100644 index 0000000..a6a6b8c --- /dev/null +++ b/scripts/observability.py @@ -0,0 +1,33 @@ +"""Initialize a deployment telemetry identity without enabling any Peer or replacing it.""" + +import json +from pathlib import Path +import sys +import uuid + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from psycopg.types.json import Jsonb +from app.database_contract.connection import database_connection +from app.schemas.observability import CONFIG_KEY, CONFIG_SCHEMA, ObservabilityConfig + + +def main() -> None: + candidate = ObservabilityConfig(deployment_id=uuid.uuid4()) + with database_connection() as connection: + connection.execute( + "INSERT INTO inkcre.configs (key, schema, value) VALUES (%s, %s, %s) " + "ON CONFLICT (key) DO NOTHING", + (CONFIG_KEY, CONFIG_SCHEMA, Jsonb(candidate.model_dump(mode="json"))), + ) + row = connection.execute( + "SELECT schema, value FROM inkcre.configs WHERE key=%s", (CONFIG_KEY,) + ).fetchone() + if row is None or row[0] != CONFIG_SCHEMA: + raise ValueError("Existing observability configuration has an unexpected schema") + config = ObservabilityConfig.model_validate(row[1]) + print(json.dumps({"deployment_id": str(config.deployment_id), "enabled": False})) + + +if __name__ == "__main__": + main() From 60de51296c12d9f23b24dbff638b8c7be2fe3de3 Mon Sep 17 00:00:00 2001 From: Lan_zhijiang Date: Sat, 3 Oct 2026 18:30:06 +0800 Subject: [PATCH 3/5] =?UTF-8?q?docs:=20=E8=AE=B0=E5=BD=95=E8=A7=82?= =?UTF-8?q?=E6=B5=8B=E5=9F=BA=E5=BB=BA=E4=BB=BB=E5=8A=A1=E5=8C=85=E4=B8=8E?= =?UTF-8?q?=E6=9C=AC=E5=9C=B0=E5=8F=8A=20Cloud=20=E9=AA=8C=E6=94=B6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../ai-instrumentation-notes.md | 39 + .../cells/contracts-foundation.md | 17 + .../cells/infra-foundation.md | 13 + tasks/observability-foundation/decisions.md | 107 ++ tasks/observability-foundation/design.md | 147 ++ .../design/hub-contract.patch | 107 ++ .../design/hub-review.md | 20 + .../design/shared-contract.md | 75 + .../experiments/.gitignore | 2 + .../experiments/README.md | 87 + .../experiments/ai-projection.py | 168 ++ .../experiments/ai_otlp_probe.py | 557 +++++++ .../experiments/application-probe.py | 197 +++ .../experiments/candidate-types.py | 73 + .../experiments/carrier-capacity.py | 48 + .../experiments/client-browser-result.json | 32 + .../experiments/client-browser.mjs | 191 +++ .../experiments/client-decode-protobuf.py | 39 + .../experiments/client-implementation.md | 85 + .../experiments/client-real-relay-result.json | 55 + .../experiments/client-real-relay-verify.py | 115 ++ .../experiments/client-real-relay.mjs | 72 + .../experiments/client-settings.png | Bin 0 -> 59502 bytes .../experiments/cloud-acceptance.md | 11 + .../experiments/cloud-probe.py | 99 ++ .../experiments/cloud-query.py | 182 ++ .../experiments/collector.yaml | 51 + .../experiments/convergence-20261003.md | 75 + .../experiments/database-client.mjs | 44 + .../experiments/database-lab.py | 98 ++ .../experiments/database-probe.py | 173 ++ .../evidence/application-probe.json | 24 + .../experiments/evidence/candidate-types.json | 7 + .../experiments/evidence/cloud-probe.json | 22 + .../experiments/evidence/cloud-query.json | 158 ++ .../carrier-capacity.json | 8 + .../database-client-result.json | 7 + .../convergence-20261003/database-probe.json | 82 + .../convergence-20261003/expected.json | 10 + .../convergence-20261003/packet-check.json | 27 + .../convergence-20261003/resource-final.json | 67 + .../convergence-20261003/stack-direct.json | 789 +++++++++ .../convergence-20261003/stack-grafana.json | 789 +++++++++ .../convergence-20261003/stack-log-input.json | 72 + .../convergence-20261003/stack-state.jsonl | 5 + .../convergence-20261003/stack-stats.jsonl | 5 + .../convergence-20261003/ui-observation.json | 31 + .../evidence/foundation-probe.json | 105 ++ .../evidence/migration-roundtrip.json | 28 + .../openobserve-20261001/ai-projection.json | 419 +++++ .../openobserve-20261001/expected.json | 10 + .../openobserve-20261001/failure.json | 105 ++ .../openobserve-20261001/propagation.json | 17 + .../openobserve-20261001/readback.json | 568 +++++++ .../evidence/runtime-probe-metrics-only.json | 68 + .../experiments/evidence/runtime-probe.json | 68 + .../cloudflare-free-check.json | 10 + .../opt-in-packet-check.json | 23 + .../sentinel-saas-20261003/packet-check.json | 23 + .../resource-final.json | 55 + .../sentinel-input.json | 322 ++++ .../sentinel-saas-20261003/sentinel.json | 605 +++++++ .../evidence/tempo-20261003/tempo-after.json | 367 ++++ .../evidence/tempo-20261003/tempo-before.json | 367 ++++ .../evidence/tempo-20261003/tempo-input.json | 256 +++ .../evidence/tempo-20261003/tempo-metrics.txt | 1477 +++++++++++++++++ .../tempo-20261003/tempo-search-after.json | 205 +++ .../tempo-20261003/tempo-search-before.json | 205 +++ .../evidence/tempo-20261003/tempo-stats.jsonl | 2 + .../experiments/failure.py | 115 ++ .../experiments/foundation-probe.py | 518 ++++++ .../experiments/implementation-db.py | 51 + .../experiments/lab.py | 150 ++ .../experiments/legacy-consumers.mjs | 29 + .../experiments/legacy-consumers.py | 46 + .../experiments/migration-roundtrip.py | 365 ++++ .../experiments/package-lock.json | 99 ++ .../experiments/package.json | 10 + .../experiments/propagation.mjs | 36 + .../experiments/propagation.py | 116 ++ .../experiments/readback.py | 115 ++ .../experiments/relay-lab.py | 103 ++ .../experiments/runtime-probe.py | 460 +++++ .../experiments/sentinel-probe.py | 152 ++ .../experiments/sentinel-saas-20261003.md | 67 + .../experiments/signals.py | 117 ++ .../experiments/stack-lab.py | 370 +++++ .../experiments/stack-probe.py | 174 ++ .../experiments/tempo-lab.py | 126 ++ .../experiments/tempo-probe.py | 116 ++ .../experiments/tempo.yaml | 17 + tasks/observability-foundation/inquiry.md | 60 + tasks/observability-foundation/packet.md | 36 + tasks/observability-foundation/task-map.md | 74 + .../observability-foundation/verification.md | 53 + 95 files changed, 13962 insertions(+) create mode 100644 tasks/observability-foundation/ai-instrumentation-notes.md create mode 100644 tasks/observability-foundation/cells/contracts-foundation.md create mode 100644 tasks/observability-foundation/cells/infra-foundation.md create mode 100644 tasks/observability-foundation/decisions.md create mode 100644 tasks/observability-foundation/design.md create mode 100644 tasks/observability-foundation/design/hub-contract.patch create mode 100644 tasks/observability-foundation/design/hub-review.md create mode 100644 tasks/observability-foundation/design/shared-contract.md create mode 100644 tasks/observability-foundation/experiments/.gitignore create mode 100644 tasks/observability-foundation/experiments/README.md create mode 100644 tasks/observability-foundation/experiments/ai-projection.py create mode 100644 tasks/observability-foundation/experiments/ai_otlp_probe.py create mode 100644 tasks/observability-foundation/experiments/application-probe.py create mode 100644 tasks/observability-foundation/experiments/candidate-types.py create mode 100644 tasks/observability-foundation/experiments/carrier-capacity.py create mode 100644 tasks/observability-foundation/experiments/client-browser-result.json create mode 100644 tasks/observability-foundation/experiments/client-browser.mjs create mode 100644 tasks/observability-foundation/experiments/client-decode-protobuf.py create mode 100644 tasks/observability-foundation/experiments/client-implementation.md create mode 100644 tasks/observability-foundation/experiments/client-real-relay-result.json create mode 100644 tasks/observability-foundation/experiments/client-real-relay-verify.py create mode 100644 tasks/observability-foundation/experiments/client-real-relay.mjs create mode 100644 tasks/observability-foundation/experiments/client-settings.png create mode 100644 tasks/observability-foundation/experiments/cloud-acceptance.md create mode 100644 tasks/observability-foundation/experiments/cloud-probe.py create mode 100644 tasks/observability-foundation/experiments/cloud-query.py create mode 100644 tasks/observability-foundation/experiments/collector.yaml create mode 100644 tasks/observability-foundation/experiments/convergence-20261003.md create mode 100644 tasks/observability-foundation/experiments/database-client.mjs create mode 100644 tasks/observability-foundation/experiments/database-lab.py create mode 100644 tasks/observability-foundation/experiments/database-probe.py create mode 100644 tasks/observability-foundation/experiments/evidence/application-probe.json create mode 100644 tasks/observability-foundation/experiments/evidence/candidate-types.json create mode 100644 tasks/observability-foundation/experiments/evidence/cloud-probe.json create mode 100644 tasks/observability-foundation/experiments/evidence/cloud-query.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/carrier-capacity.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/database-client-result.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/database-probe.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/expected.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/packet-check.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/resource-final.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-direct.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-grafana.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-log-input.json create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-state.jsonl create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-stats.jsonl create mode 100644 tasks/observability-foundation/experiments/evidence/convergence-20261003/ui-observation.json create mode 100644 tasks/observability-foundation/experiments/evidence/foundation-probe.json create mode 100644 tasks/observability-foundation/experiments/evidence/migration-roundtrip.json create mode 100644 tasks/observability-foundation/experiments/evidence/openobserve-20261001/ai-projection.json create mode 100644 tasks/observability-foundation/experiments/evidence/openobserve-20261001/expected.json create mode 100644 tasks/observability-foundation/experiments/evidence/openobserve-20261001/failure.json create mode 100644 tasks/observability-foundation/experiments/evidence/openobserve-20261001/propagation.json create mode 100644 tasks/observability-foundation/experiments/evidence/openobserve-20261001/readback.json create mode 100644 tasks/observability-foundation/experiments/evidence/runtime-probe-metrics-only.json create mode 100644 tasks/observability-foundation/experiments/evidence/runtime-probe.json create mode 100644 tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/cloudflare-free-check.json create mode 100644 tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/opt-in-packet-check.json create mode 100644 tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/packet-check.json create mode 100644 tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/resource-final.json create mode 100644 tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel-input.json create mode 100644 tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel.json create mode 100644 tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-after.json create mode 100644 tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-before.json create mode 100644 tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-input.json create mode 100644 tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-metrics.txt create mode 100644 tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-after.json create mode 100644 tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-before.json create mode 100644 tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-stats.jsonl create mode 100644 tasks/observability-foundation/experiments/failure.py create mode 100644 tasks/observability-foundation/experiments/foundation-probe.py create mode 100644 tasks/observability-foundation/experiments/implementation-db.py create mode 100644 tasks/observability-foundation/experiments/lab.py create mode 100644 tasks/observability-foundation/experiments/legacy-consumers.mjs create mode 100644 tasks/observability-foundation/experiments/legacy-consumers.py create mode 100644 tasks/observability-foundation/experiments/migration-roundtrip.py create mode 100644 tasks/observability-foundation/experiments/package-lock.json create mode 100644 tasks/observability-foundation/experiments/package.json create mode 100644 tasks/observability-foundation/experiments/propagation.mjs create mode 100644 tasks/observability-foundation/experiments/propagation.py create mode 100644 tasks/observability-foundation/experiments/readback.py create mode 100644 tasks/observability-foundation/experiments/relay-lab.py create mode 100644 tasks/observability-foundation/experiments/runtime-probe.py create mode 100644 tasks/observability-foundation/experiments/sentinel-probe.py create mode 100644 tasks/observability-foundation/experiments/sentinel-saas-20261003.md create mode 100644 tasks/observability-foundation/experiments/signals.py create mode 100644 tasks/observability-foundation/experiments/stack-lab.py create mode 100644 tasks/observability-foundation/experiments/stack-probe.py create mode 100644 tasks/observability-foundation/experiments/tempo-lab.py create mode 100644 tasks/observability-foundation/experiments/tempo-probe.py create mode 100644 tasks/observability-foundation/experiments/tempo.yaml create mode 100644 tasks/observability-foundation/inquiry.md create mode 100644 tasks/observability-foundation/packet.md create mode 100644 tasks/observability-foundation/task-map.md create mode 100644 tasks/observability-foundation/verification.md diff --git a/tasks/observability-foundation/ai-instrumentation-notes.md b/tasks/observability-foundation/ai-instrumentation-notes.md new file mode 100644 index 0000000..c6aa6e4 --- /dev/null +++ b/tasks/observability-foundation/ai-instrumentation-notes.md @@ -0,0 +1,39 @@ +# AI 首批点位与验收说明 + +此说明供主任务整合到 `docs/40-deployment/observability.md`,不是新的长期权威文档。 + +## 已实现点位 + +| 入口 | 记录 | 语义与边界 | +| --- | --- | --- | +| `AIManager.chat` / `AIManager.embed` | `ai.chat` / `ai.embed` span;操作、部署内 model/provider ID、dialect、配置中的 model identifier | 所有共同 AI 调用均覆盖,包含非 Agent 和 embedding;不读图、不修改 `AssistantMessage` 等业务合同。模型 identifier 限 ASCII 标识字符且最多 128 字符,供应商任意返回的 model/id 不复制。 | +| OpenAI-compatible response、Alibaba stream | input/output tokens、缓存输入与推理输出 token、固定结束原因;Alibaba 首块耗时 | 在 SDK 实际解析 response/chunk 后记录。流式 `choices=[]` 的最后 usage 块仍采集,后到总量替换前值,每次调用只累计一次。 | +| Agent Thread | `agent.turn` → `agent.step` → 模型和并发 `agent.tool` | 使用 thread UUID、turn 和 model-call 序号建立关系;并发工具共享 step 父 span。只对已绑定 code-owned Tool ID 记录名称;provider ToolCall ID 使用 SHA-256 摘要关联,避免任意 ID 携带正文。 | +| info-base `retrieve` Tool | `inkcre.retrieval.candidates` 结构化事件 | 逐成功 lexical/semantic 分支记录候选 block/relation ID 与候选数;不记录 query、excerpt、label 或得分。候选只是召回结果。 | +| info-base `get_entities` Tool | `inkcre.entity.read`,`read.kind=persisted` | 只记录实际非空返回的 block/relation ID 和数量。不存在的请求实体不算读到;随机读取同样按真实返回记录。 | +| Resolver Tool 成功 invoke | `inkcre.entity.read`,`read.kind=resolved` | 成功方法调用及结果投影后记录目标 block ID;不记录动态方法名、参数、schema 或结果。此点位不声称覆盖 Resolver 内部所有传递读取。 | +| `AgentQuerySink.execute` 选取结果 | `inkcre.agent.query.result` | 在最后一个已闭合成功提交被选入 `Job.state.result` 后记录 Job ID、thread UUID、最终引用 ID 与数量。不是每次 submit 都算最终结果,也不声称此时 Job 的最终持久化已经成功。 | + +GenAI 属性固定映射到 `semantic-conventions-genai` 修订 `e07f4ebacb08f56db8c4c882d117720333fbca04`,映射位于 AI owner 的 `telemetry.py`。span 名称保持固定,动态 model、Tool 和业务 ID 不作为指标维度。token 计数使用自有 `inkcre.ai.token.usage` counter,不冒充语义约定中不同聚合类型的同名指标。 + +数字 usage 接受非负 int64,真实零保留,缺失、负值、布尔或溢出值均不写入 token 计数;input/output 的 `source` 分别为 `provider` 或 `unavailable`。缓存和推理维度只记录供应商明确返回的已知值,不额外加到 input/output 总量。缺失字段另计 `inkcre.ai.usage.missing`。不估算成本、不创建价格服务,也不将部分 token 和称作完整用量。 + +原始 finish reason 只允许 `stop`、`length`、`tool_calls`、`content_filter`、`function_call`,其他值映射为 `other`,每次最多保留 16 项。基础模式不复制模型输入输出、provider config、异常文本或 stack。现有 `agent_debug`、PG/Logtail 内容行为不变,不能据此宣称整个部署没有内容日志。 + +最终引用最多导出前 128 项中有效的正 int64 ID,并记录完整 `reference_count` 和未导出数量 `unreported_reference_count`。业务结果自身不截断、不增加验证或持久化模型。最终引用由模型提交,记录关联不验证实体存在,更不证明答案确实基于该实体或答案正确。 + +首批来源点位限上述 Tool 和 Agent Query Sink;图遍历 Tool、直接非 Agent 检索 API、Resolver 内部依赖读取尚不属于来源关联覆盖范围。AI 共同执行 span 本身已覆盖全部经 AIManager 的模型调用。这些 trace/log 会采样和过期,不替代答案或图谱变更的长期证据快照。 + +## 已执行验证 + +`pdm run python tasks/observability-foundation/experiments/ai_otlp_probe.py` 使用本地 HTTP 合成 provider、真实 OpenAI SDK 解析、真实生产 Tool/Resolver/Thread/Sink 和 OTel SDK/exporter,并在首次 OTLP HTTP 入口解码三信号。数据库读取和 Agent 定义装载使用合成领域对象隔离;没有真实供应商或 PostgreSQL 端到端验收。 + +实验显式设置 `INKCRE_ENV_FILE=''`,只配置本地三信号接收端,并使用合成数据库/JWT/provider 参数;不读取工作区 Grafana 或外部 provider 凭据。最终运行获得 13 次 provider HTTP 请求(其中 1 次验证关闭遥测)、31 个 span、4 个来源事件,并通过以下观察: + +- 关闭遥测时在既有第三方 span 内执行真实 Alibaba HTTP 流式调用,不改写该 span 属性、不替换外部 current context。 +- unknown、零、部分和负 usage,以及 embedding 与流式末尾 usage 分别符合合同;中途和末尾相同总量不会重复累计。 +- 并发工具有共同 step 父 span、时间重叠,工具错误结果与取消传播不变。 +- 同一 Agent 旅程中候选 block `(11,12)`、实际读取 block `(11)` 与 relation `(21)`、Resolver 目标 block `(11)`、最后选中引用 relation `(21)` 分开记录,四事件在同一 trace。 +- 同批两次成功 submit 只记录最后一个最终结果;正文、query、工具输入输出、provider 回显 ID/model、异常消息与答案的统一 canary 在首次三信号出口均不存在。 + +AI/Agent Ruff、格式和 pyrefly、数据库 import/ownership 边界检查通过;既有 `tests/agent/test_debug_trace.py` 4 项通过。真实供应商行为、真实检索数据库与已获准 preview/SaaS 是剩余验收层次。 diff --git a/tasks/observability-foundation/cells/contracts-foundation.md b/tasks/observability-foundation/cells/contracts-foundation.md new file mode 100644 index 0000000..783156c --- /dev/null +++ b/tasks/observability-foundation/cells/contracts-foundation.md @@ -0,0 +1,17 @@ +# C-G1:共享契约基线 + +状态:2026-10-03 业务契约已返回 G1;D10 补齐默认关闭、本地显式 opt-in、PG 保留及 mixed Peer 语义;按信号出口与 carrier 形状保持。主 Agent 拥有契约设计与兼容判别;授权表面仅为任务内提案、只读调查和 disposable 数据库实验。未修改 Hub、共享挂载或运行源码。 + +## 返回 + +[共享契约形状](../design/shared-contract.md)已经确定身份/config、两列 carrier、512-byte 约束、SDK 超限 state 省略、PG 日志兼容与协调升级。[Hub patch](../design/hub-contract.patch)是四文件未应用差异,[评审说明](../design/hub-review.md)记录归属和交付边界。 + +标准 Python/JS propagator、真实旧 Python repository、TS JobManager/DBAPIClient 经 PostgREST 均有证据。缺少 carrier 的旧创建正常,已有 carrier 在 claim/close 后保留;模拟新 head 导致旧 readiness 拒绝,因此不宣称任意旧新版本滚动兼容。SDK 注入合法 tracestate 可超过 512 bytes,应用持久化政策与 W3C 上限明确区分。详见[实验报告](../experiments/convergence-20261003.md)。 + +## 下一消费方与不变量 + +C 的下一消费者是 Hub 发布和 core-py 数据库协议/各语言采集实现,具体顺序由[工作地图](../task-map.md)拥有,不另建控制入口。正式 migration、生成投影、初始化命令与跨 Peer 运行仍须实施验收;实验 DDL 不是正式发布物。 + +Job 领取、取消、timeout、终态和未知结果语义不由 Trace 决定;没有上下文的 Job 仍可执行。部署 ID 不提供权限、不从 secret 派生、不由各 Peer 随机各生成一份。发现共享 truth 冲突或新持久成员需求时返回主任务,不扩展成通用执行/遥测存储。 + +本单元完成不清理任务包;Q1/Q2 的生产内容与预算输入不阻挡已完成的契约基线,也没有因此被默认为已确认。 diff --git a/tasks/observability-foundation/cells/infra-foundation.md b/tasks/observability-foundation/cells/infra-foundation.md new file mode 100644 index 0000000..e3464f5 --- /dev/null +++ b/tasks/observability-foundation/cells/infra-foundation.md @@ -0,0 +1,13 @@ +# I-G1:基建可行性 + +状态:2026-10-03 Sir 已同意 Grafana Cloud Free;按 D10 默认关闭、PG 保留与 vendor-agnostic 进行实现准备,云端准入仍未运行。主 Agent 拥有任务内提案及独立合成实验;运行接入和数据生命周期由部署 owner 负责。 + +既有 OpenObserve 三信号、Tempo 重启/Link 和五组件 API/Grafana UI 闭环见[前轮](../experiments/README.md)与[数据库/三信号报告](../experiments/convergence-20261003.md)。这些实验有效,但默认自建结论被 Sir 明确的 serverless、scale-to-0 与 SaaS 偏好替代,不再从608MiB快照推断实际运维成本合适。 + +本轮证实 OO 原始 -1 与来源标记可保留,自动派生 total/cost 仍需按来源解释。官方 SaaS 能力和费用见[修订报告](../experiments/sentinel-saas-20261003.md)。已选 Grafana Cloud Free 为按需开启的后端,OpenObserve Cloud 因零费用约束暂缓;Cloudflare 原生能力按实际 Unit 补充。不恢复默认常驻 Collector,不启动更多产品大全式比较。PostHog 原生三信号已有,但当前成熟度及 AI/普通 Trace 关联限制使其不成为唯一基建首选。 + +返回消费方是[整体设计](../design.md)与[工作地图](../task-map.md)。剩余准入是目标账户的必要因果/AI查询、导出、三信号、浏览器入口、有界 flush 与费用约束;本地实验不能证明这些云端行为。源端来源/值边界、真实业务故障隔离和覆盖仍由 G2/G3 验收。 + +所有实验容器和隧道已停止,合成卷随父任务保留,不动 SVC 数据库。标准传播、Job carrier、协调升级及业务 authority 保持,不因后端修订重做已有效的数据库判别。 + +默认关闭与 PG 保留须由 V0/V3 证明;仅配置 endpoint/token 不能开启,新开关不切换 logging_backend。供应商相关配置/查询留在部署与消费侧,关闭/开启混用不改变 Job 结果。Compose 现有 none fallback 的对齐属于待实现项,不能以任务包完成冒充已修复。 diff --git a/tasks/observability-foundation/decisions.md b/tasks/observability-foundation/decisions.md new file mode 100644 index 0000000..ab3bd13 --- /dev/null +++ b/tasks/observability-foundation/decisions.md @@ -0,0 +1,107 @@ +# 任务决定与重开条件 + +现行 Grafana Cloud Free 选型、vendor-agnostic 与默认关闭/PG 保留由 D10 拥有;零费用及 Cloudflare 范围由 D9 保留。D8 的 SaaS/SDK 直发方向保留,OpenObserve Cloud 优先顺序被 D9 替代。D1/D4/D6/D7 中的默认常驻 Collector、自建配方及最终后端收敛结论已被 D8 替代,保留它们作为历史决策。这里记录任务级选择及其重开条件;尚未成为 Hub 契约或运行行为的提案,不能因写入本文件而视为已经交付。 + +## D1:统一采集协议,保留后端替换能力 + +- **状态与权威**:Sir 于 2026-10-01 认可架构草案、方向与技术选择,允许继续推进。 +- **选择**:以 OpenTelemetry/OTLP 为采集与传输基础,部署自有 Collector 汇聚;OpenObserve 单机作为首个验证后端,业务采集点不依赖其专有 SDK。 +- **原因**:同时覆盖多 Peer 的日志、指标和 Trace,并把更换存储/查询产品的成本限制在消费侧。 +- **后果**:必须验证数据导出、字段保留、Span Links 和必要查询;不能把“支持 OTLP”当作完整可迁移性证明。已有合成摄取、查询和中断证据;生产适用性仍未确认。2026-10-01 的 AI 缺失值实验触发后端重开,见 D4。 +- **重开条件**:实际资源超预算、关键查询或导出能力缺失,或者发现已有更合适的运行平台。先调整后端,不轻易改变采集协议。 + +## D2:长期正确通过边界与证据体现 + +- **状态与权威**:Sir 于 2026-10-01 明确要求长期正确;以下是主 Agent 对已认可架构的工程落实。 +- **选择**:保留 Peer 平等、数据库业务 authority 和现有 Job 语义;诊断遥测与业务结果证据各有保存责任。迁移必须覆盖旧消费者,并以独立查询、故障注入和实际业务结果验收。 +- **原因**:短期 Trace 的采样、过期及丢失不能决定业务状态或长期证据是否存在;只替换日志后端也无法证明跨 Peer 关联正确。 +- **后果**:先做贯穿 Python、浏览器、异步 Job 与 AI 的集成切片,再扩面;不通过跳过兼容、数据边界或恢复验证来缩小任务。 +- **重开条件**:产品明确改变 Peer/owner 模型,或要求无损审计、不可篡改证据、执行恢复等新的承诺。该变化需单独决策,不能借“长期正确”自行推导。 + +## D3:当前可推进的范围与未授予的含义 + +- **状态与权威**:Sir 已要求扩展任务包并继续推进;主 Agent 据此继续契约设计、只读调查与隔离合成实验。 +- **选择**:同一部署内汇聚作为现行产品模型下的工作基线,基础元数据与原始内容分开控制。没有明确内容范围前,实验使用合成数据。 +- **原因**:当前一个部署只有一个 owner,尚无产品租户模型;原文保存会改变数据副本、读者和删除责任。 +- **后果**:预算和内容问题只阻挡依赖它们的生产选型或内容启用,不阻挡任务包、协议设计和合成实验。2 vCPU/4 GiB、保留天数、永久证据强度都未成为正式要求。 +- **重开条件**:Sir 指定跨部署集中运营、原文/长期证据范围、部署环境或资源上限。 + +架构认可不等于 Git 提交、推送、生产发布或现有私人内容导出的授权。实际授权在执行对应动作前按会话与仓库规则判断。 + +## D4:否决当前 OpenObserve,选定分立后端 + +- **状态与权威**:2026-10-03,主 Agent 根据可重复实验及 advisor 复核作出的候选判断,属于 D1 已约定的条件验证;并非 Sir 已批准新的生产后端。 +- **选择**:OpenObserve 1.0.4 不作为当前统一后端;沿 Grafana/Tempo、Loki、Prometheus 方向验证。随后五组件合成三信号与 Grafana UI 闭环通过,选择该组合进入实现;尚未认可生产容量或正式部署。 +- **原因**:OpenObserve 在摄取时把缺失 usage/cost 补零,使 unknown 与真实零不可区分;不能靠查询还原。Tempo 重启后保留了样本的字段存在性、类型和值及 Link ID/tracestate/属性。 +- **后果**:不增加影子字段、供应商修复层或自研存储;不同时扩大产品比较。Tempo 历史 Link flags 丢失是保留的已知限制,完整 OTLP 保真未通过。SDK、Collector 和 Job carrier 继续使用完整标准传播;查询缺失的 flags 不代表原始未采样,不能用于恢复传播、重采样或推断完整性。 +- **重开条件**:后续消费者需要历史 flags,日志/指标的字段映射破坏当前语义,全栈资源或操作成本不合预算,或者新版本改变以上行为。需要时只重验受影响能力。 + +[实验记录](experiments/README.md)拥有版本、输入、查询、资源和原始证据。当前协议出口与三信号/查询 UI 闭环已有证据;历史数据迁移、真实业务与生产运行仍未完成。 + + +## D5:固定公共形状,采用协调升级 + +- **状态与权威**:2026-10-03,主 Agent 根据真实 disposable PG/PostgREST 与 SDK 容量实验定稿,advisor 复核未发现阻挡实现的设计问题。 +- **选择**:两列 nullable carrier、各 512 bytes;SDK 超限 optional state 在写入前整体省略。部署标识复用一个现有 configs 记录,由部署启用步骤一次初始化;具体字段与 config key/schema 在[共享契约](design/shared-contract.md)。 +- **原因**:旧 Python/TS 实际创建与 claim/close 保持兼容,但旧 exact-head readiness 确实拒绝新 head;SDK 也不会自动把 tracestate 限为 512 bytes。 +- **后果**:协调升级、匹配 runtime/schema、旧客户端刷新,不增加静默删字段重试或滚动兼容框架。不得把实验 DDL 当正式 migration。 +- **重开条件**:必须零停机混跑、增加新的持久传播成员或既有 config authority 发生改变。 + +## D6:独立配方和有限采集来源 + +- **状态与权威**:主 Agent 在既有交付物盘点、五组件闭环及内容边界复核后的工程选择;用户未要求创建新的平台仓库。 +- **选择**:core-py 分发可独立运行的 Compose 配方,部署 owner 运行。基础远端采集只接明确来源和字段,专用结构化事件 logger、受控 span;自动 instrumentation 按最终记录准入,不默认桥接任意日志。 +- **原因**:已有仓库负责自托管 Compose;新仓/控制服务没有现成收益。任意异常/SQL/SDK 日志包含原文的可能性不能靠只删 headers 或通用正则证明消除。 +- **后果**:可信 owner 用 Grafana Editor/Explore;Viewer 不承诺交互诊断。原始 PG writer 属于内容模式。首版依旧使用标准 SDK/认证/队列能力,不自建 OTel 数据重写框架。 +- **重开条件**:独立 infra 产品出现、必须给只读角色完整 Explore 能力、跨 owner 运营、或准入自动 instrumentation 无法满足最终记录边界。 + +## D7:关闭设计阶段,保留实施验收 + +- **状态与权威**:Sir 要求“直到方案完全收敛”;2026-10-03 G1 判别完成,独立 advisor 建议结束选型实验。 +- **选择**:方案按单 deployment、2 CPU/4 GiB 合成试点、AI 诊断/来源关联与原文默认关闭进入实现准备。以上是设计假设;实际预算和历史快照承诺未由 Sir 确认。 +- **后果**:不再无限扩展候选比较。G2 验真实 Peer/浏览器/AI/故障/出口,G3 验活跃覆盖/容量/运维,V10 根据内容承诺验收;整个基建父任务仍活跃。 +- **重开条件**:实际范围或预算否定假设,或者实现判别否定关键机制。不能把未执行验收改称已完成。 + +## D8:按 serverless 与 scale-to-0 修订为 SaaS 优先 + +- **状态与权威**:2026-10-03,Sir 明确否定自建五组件的常驻成本,偏好轻量 OpenObserve,并建议利用 PostHog 等 SaaS。该输入重开 D7 的预算假设,替代 D1/D4/D6 的默认自建结论。独立 advisor 已就新约束、最低必要语义和短生命周期进行复核。 +- **选择**:默认标准 SDK 直发 SaaS,不要求常驻 Collector。首先验证 OpenObserve Cloud,Grafana Cloud Free 为明确备选;PostHog 有原生三信号与 AI 分析,但 tracing Beta、metrics Alpha 和关联限制使其暂不成为唯一基建首选。候选顺序是工程推荐,不代表用户已经开通或批准采购。 +- **原因**:运维、闲置固定费和应用生命周期比单次内存快照更能衡量轻量。OTLP 与业务身份提供替换能力,不要求自建数据平面。当前必要字段可消费即可,不要求后端逐位保真。 +- **缺失值决定**:`-1` 可以是消费端受控的 unknown 编码,不能当实际 token/cost 聚合。源端标准字段缺失省略;对于会补值的后端,允许最少的逐字段来源元数据并配套查询,撤销“任何额外标记都是影子真相”的过度约束。不为此自研存储或通用修复框架。部分未知的完整 total 仍未知,费用仅统计已知部分并展示缺失数。 +- **配置影响**:尚未发布的 v1 配置从单一 `otlp_http_endpoint` 改为可选 `otlp_http_endpoints`,按 traces/logs/metrics 保存无凭据的完整 URL,映射标准 per-signal exporter 配置。供应商不同路由不构成必须上 Collector 的理由;凭据仍在运行/连接配置中。 +- **验收与停止比较**:用一个目标账户验证三信号、Job 因果、AI unknown/zero/partial、浏览器入口、导出和用量边界。通过必要诊断能力即进入 G2,失败才转备选;不因为当前不用的字段或缺少专用 AI UI 无限扩大选型。 +- **保留与重开**:D2/D3/D5 的业务 authority、内容范围、Job carrier 和协调升级保持;原自建实验仍是历史证据。只有不可接受的真实费用、必要语义/查询缺口、入口边界或生命周期开销才重开。目标 Cloud 行为、实际负载/预算与长期证据承诺没有被假定为已验证。 + +## D9:新增观测服务费用为零,先验 Grafana Cloud Free + +- **状态与权威**:2026-10-03,Sir 询问 Cloudflare 服务,并明确暂时无法使用收费服务。该约束立即替代 D8 的 OpenObserve Cloud 首验顺序;不把14天试用或低单价当免费。 +- **选择**:首先验证 Grafana Cloud 的实际 Free 计划,用标准 OTLP 接受多 Peer 三信号;无需自建底层 Tempo/Loki/Prometheus,也无需默认 Collector。Cloudflare 原生观测覆盖在其平台运行的 Unit,按内容/关联准入后可导出到同一后端;AI Gateway 只在模型代理需求成立时单独评估。OpenObserve Cloud 暂缓,PostHog Free 保留能力资料,当前不并行建设第二套系统。 +- **原因**:免费完整 SaaS 比拼接 Workers、D1/R2、Analytics Engine 成为自研观测平台更符合低维护目标。Cloudflare 公开文档尚不足以证明通用外部 OTLP 三信号存储入口;Analytics Engine 的采样也不能保证单条诊断记录可找回。 +- **费用与生命周期**:不采购、不升级 Pro、不启用付费附加项、不以试用能力作为验收基线;预算耗尽时遥测可拒收/丢失,业务照常。应用发送、转发与平台采集均受额度约束,不为观测新增常驻实例。已有业务/模型推理成本不因观测免费消失。 +- **最小准入**:目标账号确认 Free 及三信号限额,合成样本读回 Job/Trace/Link 与 AI unknown/zero/partial,确认消费 UI、样本导出、浏览器凭据边界和有界 flush。指标按类型/基数验,Cloudflare 自带导出不支持 metrics 的缺口逐 Unit 如实列出。 +- **时间与残余**:Cloudflare 12月1日新定价尚未生效;AI Gateway 新旧客户以9月24日区分,见[官方研究](experiments/sentinel-saas-20261003.md)。当前只做只读研究和任务包修订,目标账号实验仍未执行。 +- **重开条件**:实际必要语义/接入不能满足,免费额度不足以支持有用诊断,或 Sir 改变费用/保留要求。先降低可选遥测量并准确呈现缺口,不能擅自付费或放宽业务/证据契约。 + +## D10:用户选定 Grafana Cloud Free,显式 opt-in,默认保留 PG 日志 + +- **状态与权威**:2026-10-03,Sir 明确同意 Grafana Cloud Free,强调 vendor-agnostic、新可观测性按需开启,默认维持现有 PostgreSQL 日志。后端从工程推荐转为用户已选,实际 Cloud 验收仍待执行。独立 advisor 支持单一本地开关、关闭态不参与新增遥测与 mixed Peer 契约。 +- **默认行为**:新遥测关闭,现有 stdout、logging_backend、PG writer 的级别/上下文门控/队列/关闭顺序和 Job 日志页面保持。启用 OTLP 也不自动停 PG;撤销先前“新查询通过后退役旧 writer”和“保留 PG 需另启 legacy 内容模式”的计划。现有显式 logging_backend 覆盖继续有效。 +- **配置选择**:每个 Peer 使用本地 telemetry_enabled,缺省 false;Python 拟用 OBSRV__TELEMETRY_ENABLED,浏览器用连接本地布尔配置。共享 configs 只给身份/目的地,不提供总开关,endpoint/token 不等于开启。重启/连接重新初始化生效,首版不做热切换。 +- **关闭语义**:不初始化新增 SDK/provider/队列/instrumentation,不新增传播或 carrier 捕获;新建 Job carrier 默认 NULL,已有 carrier 经 claim/close 保留。开启端可以执行 NULL carrier Job,关闭端可以执行有 carrier Job;链路缺口允许,业务与准入规则不变。关闭不清理历史,不回填或补发此前未采集记录。 +- **vendor-agnostic**:标准 SDK/OTLP、标准 ID 与公共语义进入业务采集;供应商 endpoint/认证/查询/面板在部署或消费侧。更换后端不改业务采集代码、Job carrier 或 schema。查询和历史迁移仍有成本,不为消除全部差异增加通用插件/转换框架。 +- **已观察差异**:libs/obsrv/setting.py 和 .env.example 默认 PG;docker-compose.yml 未配置时使用 none。实现时将 Compose fallback 对齐用户要求,保留显式 none/logtail 等配置;当前尚未修改它,不能宣称所有启动方式都已默认 PG。 +- **必要验收**:默认关闭、端点/凭据存在但未开启、显式开启、出口故障、再次关闭的 PG/UI/OTLP 行为;开关组合的跨 Peer Job 与字段保留;仅改出口配置即可替换后端。新出口的内容策略不倒推覆盖旧 PG 日志内容,不能声称整个部署默认无原文。 +- **边界**:本轮修订任务包及未应用 Hub patch,应用实现、目标账户与正式发布仍未执行;继续零新增观测服务费,不启动收费功能。 + + +## D11:授权实施与首批交付 + +2026-10-03,Sir 明确授权实施,并允许创建分支、提交、推送和 draft PR(包括 client-web);随后创建 Grafana stack 并提供本地写入与 Viewer 配置。Hub 源已先提交推送,两 Spoke 的引用各自单独提交,应用实现按该共享契约进行。授权不扩展为合并、生产发布或付费服务。 + +首批采用标准 OTLP/HTTP protobuf。真实 browser relay 读回发现通用 ProtoJSON 会把 OTLP hex Trace/Span ID 按 base64 解码;HTTP 200 不能证明正确。改为客户端官方 protobuf exporter 与服务端 protobuf-only,避免应用自研协议解码器。JSON 返回 415,最终验收比较 ID、Link 与事件上下文原值。 + +SDK 原生内部指标通过 `OTEL_PYTHON_SDK_INTERNAL_METRICS_ENABLED=true` 的进程环境启用;Compose 默认 true 且允许覆盖,原生启动显式注入。它不绕过遥测总开关,View 限定内部指标维度,并禁止过滤后 exemplar 携带被移除的属性。缺少 metrics 出口或关闭此开关时没有原生丢弃计数;计数自身丢失也不能推断零丢弃。避免自定义队列或私有 SDK 工厂。 + +独立 advisor 认为这些选择不阻挡首批 draft;候选数据库类型、真实平台关闭行为及部署覆盖仍需各自准入。原任务包状态随实际实现更新,不以 draft 或本地测试替代生产通过。 + +真实 Cloud 验收进一步否定固定一秒网络超时:原生 TLS 首次往返超过一秒。使用标准公共/per-signal timeout(默认十秒、最多三十秒),relay 总预算为该信号 timeout 加两秒。Grafana Logs 返回204,转发识别200/204并以标准空protobuf返回;实际API读回确认已存储。SDK原生队列计数有并发误差,只作诊断,不当精确损失或费用账本。 diff --git a/tasks/observability-foundation/design.md b/tasks/observability-foundation/design.md new file mode 100644 index 0000000..ca7020d --- /dev/null +++ b/tasks/observability-foundation/design.md @@ -0,0 +1,147 @@ +# InKCre 可观测性基建设计 + +状态:2026-10-03 根据 Sir 明确的 serverless、scale-to-0、SaaS、当前零新增观测服务费、vendor-agnostic 与默认关闭要求修订。标准采集、业务契约已收敛;部署拓扑已改为 SaaS 优先,目标账户准入尚未验证。本文件拥有整体设计,[共享契约](design/shared-contract.md)拥有字段和兼容形状,[工作地图](task-map.md)拥有实施顺序,[验收表](verification.md)区分实验通过与尚未实现。设计定稿不等于 Hub 契约已生效或整套基建已经交付。 + +显式开启后采用 **OpenTelemetry SDK/OTLP → SaaS 托管的采集、存储与查询**,不默认新增常驻 Collector、数据库或监控主机。短生命周期应用停机后,观测基建不能要求它保持在线;闲置固定费用、采集端开销和实际维护责任共同决定轻量程度。Sir 已同意 Grafana Cloud Free 作为首个可选后端;OpenObserve Cloud 需要收费,当前暂缓。选型已获同意,目标账户与实际云端接入尚未验收。PostHog 作为有原生 OTLP 三信号和 AI 分析的候选保留,但其 tracing Beta、metrics Alpha 与分析关联限制需纳入取舍。 + +五组件及真实旧数据库消费者实验仍提供协议和查询证据,不再证明自建方案符合 Sir 的部署目标。一个 deployment 仍属一个 owner,各 Peer 平等参与。AI 首阶段交付运行诊断及结果来源关联,原文默认关闭;长期快照由结果 owner 按明确承诺保存。 + +## 要回答的问题 + +新增遥测开启并覆盖相关 Peer 后,首阶段应能回答:一次操作在哪个 Peer 执行,等待在哪里发生,哪个模型或工具失败,最后的答案或图谱写入使用了哪些输入,以及观测系统自身是否正在丢数据。 + +[现状证据](inquiry.md) 已确认 Python 与浏览器都是 Job 的生产者和执行者,旧 Job UI 依赖 `job.`,AI adapter 尚未保留 usage。因此方案必须覆盖对等执行、兼容迁移和 provider 边界,不能只增加 core-py HTTP middleware。 + +## 默认行为与按需开启 + +现有 PostgreSQL logging 是保留能力。新遥测开关与 `logging_backend` 独立,默认关闭;安装依赖、升级 schema、保存共享 endpoint 或放入凭据都不能自动启用。源码及 .env 模板默认 PostgreSQL,Compose 的未配置 fallback 当前为 none;实现将该 fallback 对齐为 PostgreSQL,同时保留显式 none/logtail 等原有覆盖。PG 写入的上下文门控、级别、有界队列与关闭次序继续保持,不把“默认 PG”扩大成所有日志无条件落库。 + +每个 Peer 用本地 `telemetry_enabled=false` 表达 opt-in。进程 Peer 从运行配置读取,浏览器从该部署连接的本地配置读取;共享 `inkcre.observability.v1` 只提供身份和目的地。首版在进程启动或浏览器连接重新初始化时生效,不承诺运行中热切换。关闭时不初始化新增 provider、processor、exporter、后台队列或自动 instrumentation,也不新增 Trace Context 注入/提取和 Job carrier 捕获;保留现有日志关联机制。启用所需配置不足则清楚报告新遥测初始化失败,不改变业务 readiness 或旧日志路径。 + +开启后 PG 继续记录原有日志,OTLP 只增加声明的元数据记录;不能把 PG 历史、任意日志文本或 agent_debug 内容自动转发。撤销“新查询通过后停用 PG writer”的旧计划。已有 `logs.trace_id=job.` 等字段不改成 OTel Trace ID,新增标准关联在新的遥测记录里表达。Job 页面始终提供原日志,外部诊断链接是附加能力。 + +关闭参与的 Peer 会产生明确的诊断缺口:其新建 Job carrier 为 NULL,更新已有 Job 时保留此前 carrier;开启的执行 Peer 对有效 carrier 建 Link,对 NULL 建独立 trace。不开启的执行 Peer 正常执行且不产出新 span。关闭并重新初始化不删除历史日志或 carrier,也不把缺少 span 解释成没有执行业务;详见[共享契约](design/shared-contract.md)。 + +## 采集、汇聚、存储与查询 + +以下图示仅表示显式开启的新增路径;现有 PG 日志在两种状态下均保留。 + +```mermaid +flowchart LR + P["进程 Peer / core-py / Extension"] -->|"标准 OTLP,有界发送"| S["SaaS 采集 / 存储 / 查询"] + W["浏览器 / 原生 Peer"] -->|"适合客户端的受限写入入口"| S + W -. "需要保护服务端凭据时" .-> R["受控的按请求转发入口"] + R --> S + H["平台已有日志 / 指标出口"] --> S + W -. "业务请求传播 Trace Context" .-> P + P -->|"现有日志路径保持"| L["PostgreSQL logs"] + P --> D["原有业务结果与证据引用"] + D -. "按业务 ID 关联" .-> S +``` + +进程 Peer 直接使用 SaaS 提供的标准出口;进程内 Extension 复用宿主 SDK。浏览器仅在有适合公开客户端的受限写入能力并通过 CORS/滥用边界验收时直连;否则使用既有认证下的按请求转发,优先复用现有平台能力。不能把服务端私密 ingest、查询或管理凭据写进静态 bundle。转发只负责必要的认证和路由,不增加自研遥测协议、持久队列或查询平台,也不能让浏览器 Job 的业务路径被迫经过 core-py。 + +应用使用标准 SDK、context manager、propagator、batch processor 和 exporter;我们只补 HTTP/Peer、Job/Cron、AI/Tool 及结果这些共用业务边界。基础模式导出的是允许的元数据记录:固定操作名/事件名、路由模板、状态、耗时、受控 ID、真实 usage 与异常类型。字段许可同时限制值的来源,不能把任意用户字符串装进一个获准字段。浏览器在共用 Peer/DBAPIClient 调用边界接入,不全局劫持全部 fetch。 + +不默认启用全量 FastAPI/HTTPX/SQLAlchemy 自动埋点。仅当固定版本配置或 hooks 能约束最终完成记录的 span 名称、属性、events 和 status 时,才逐个准入。最小公共 span 入口统一关闭 `record_exception` 与 `set_status_on_exception`,失败时显式设 ERROR 与异常类型;仅关前者仍可能把异常文本写入 status description。数据库先记录业务调用耗时和连接池指标,不自动导出 SQL statement/参数。无需为保留全面自动埋点构造通用 Span 重写层。[固定版本 SDK 行为](https://github.com/open-telemetry/opentelemetry-python/blob/v1.45.0/opentelemetry-api/src/opentelemetry/trace/__init__.py) + +新增 OTLP logging bridge 只接专用结构化事件 logger,body 为固定事件名;不扩大原有 root、应用/第三方 logger 的出口范围,stdout 与配置的 PG/Logtail 路径继续保持。SDK 的非法 tracestate 警告因此不进入远端基础日志。Resource 同样明确列取,不自动上传命令行或任意运行环境。既有 PG/Logtail writer 可能继续包含动态内容;新增 OTLP 的元数据承诺只覆盖新出口,不要求用户再次开启旧日志,也不能据此声称整个部署没有原文副本。 + +SDK 使用现有有界 batch、超时和有限重试。常驻进程在正常关闭时有界 flush;短生命周期任务在平台保证的执行窗口结束前 flush,测量新增延迟和计费时间,不能假定 HTTP 响应后仍能继续发送。冻结、崩溃、浏览器关闭可能丢失尾部数据,重试也可能重复;两者都不能改变业务结果。遥测不可用不阻止启动或 readyz,保留本地诊断。 + +应用指标优先 OTLP push;不为 scrape 唤醒已缩至零的实例。实例退出后的无数据不能当成 usage=0 或故障。全局 pending Job 等状态由既有持久读源或实际运行的调度责任采集,不为面板单独增加常驻轮询器。平台托管 PG/PostgREST 的可用出口按实际能力接入,不自称采到了不可见的数据库内部行为。 + +只有已证明需要宿主采集、协议转换、多目标路由或额外缓冲时才增加 Collector;复用其原生接收器、认证、重试和队列,不自行实现。跨重启积压需求同时意味着持久存储和运行成本,需有具体收益才启用。[Collector 可靠性边界](https://opentelemetry.io/docs/collector/resiliency/) + +### SaaS 选择与运行责任 + +| 部署选择 | 适用性与准入重点 | +| --- | --- | +| Grafana Cloud Free | Sir 已同意;按需开启时使用,并验证实际 Free 计划;托管三信号、非限时免费且无需信用卡,不部署 Tempo/Loki/Prometheus,不开 Pro 或付费附加项 | +| Cloudflare 原生观测 | 覆盖其平台 Unit,内容/关联准入后可导出 logs/traces 到统一后端;AI Gateway 按模型代理需求单独评估,Analytics Engine 不作唯一 Trace 库 | +| OpenObserve Cloud | 因当前零费用约束暂缓;既有缺失值实验及兼容办法仍有效,不把免费试用当长期方案 | +| PostHog Cloud | 原生 OTLP 三信号和 AI 分析均存在;tracing Beta、metrics Alpha,普通 Trace 与 AI 分析尚非统一查询/UI。当前不作为唯一基建首选 | + +[本轮判别报告](experiments/sentinel-saas-20261003.md)拥有当前官方费用、成熟度、入口差异与 `-1` 实验。正式采用前用一个实际 Free 账户确认限额、无付费依赖,并验证必要的因果查询、unknown/zero/partial、counter/histogram、浏览器写入边界、样本导出、配额和关闭。没有账户实验就不能把本地 OpenObserve 或自建 Grafana 的结果标成 SaaS 验收通过。 + +部署 owner 管理 SaaS 项目、区域、写入/读取角色、额度、保留、导出和删除。当前新增观测服务费为零,不能自动升级计划、启用收费扩展或依赖限时试用;额度不足时削减可选遥测并呈现丢失。采集费用包含应用 flush/网络成本;存储/查询费用按目标计划计量。保留期取供应商实际可用设置,不沿用自建 7/30 天假设。为额度耗尽和丢弃提供可见状态;告警和查询不得通过不断访问业务服务破坏 scale-to-0。首版不自建供应商账单服务或控制面。 + +不交付默认自建五组件配方。已有隔离配方保留为协议实验及确有自托管需求时的备选;其 2 CPU/4 GiB、608 MiB 快照与约 8 分钟 Grafana 冷启动不是生产预算依据。未来选择自建时另验冷备/恢复、磁盘回收和认证;托管方案则验配置重建、受支持导出/恢复、保留和删除,不能承诺访问或恢复供应商内部卷。 + +“避免锁定”是实现约束:业务采集不引入 Grafana 专有 SDK、项目或 datasource ID,不按 vendor 分支;身份/carrier/schema 不包含供应商标识。部署配置拥有 OTLP 端点与认证,查询/面板及诊断链接的后端语法归消费侧。标准 SDK、独立业务 ID、明确字段映射与必要记录导出构成更换后端的基础。切换可改 SDK 出口或按需要短期双发,不把一直运行 Collector 作为可替换性的前提。SaaS 账户、查询语言、面板和历史数据仍有迁移成本;优先让旧保留期自然结束,只有实际历史需求才做迁移。后端不必原样保留每个 OTLP 字段,但必要因果 ID、unknown/zero 与聚合语义必须能正确消费;历史 Link flags 或不用的自动推导字段不是单独否决项。 + +## 多 Peer 关联契约 + +以下关联行为适用于开启的 Peer;关闭或未接入 Peer 允许造成诊断缺口但必须保持业务正确。[共享契约候选](design/shared-contract.md)拥有当前字段、身份来源和旧新消费者兼容提案,尚未发布到 Hub。 + +使用 OTel Resource 区分 deployment、service、version、environment、Peer 与进程实例。部署标识在同一 owner 的 Peer 间一致;Peer 标识复用已有身份;进程实例随启动变化。尚未注册的启动日志不伪造 Peer ID,浏览器身份由其现有运行契约提供。业务 ID 作为属性,不能兼任 Trace ID。[OTel Context Propagation](https://opentelemetry.io/docs/concepts/context-propagation/) + +| 路径 | 提议的关联方式与约束 | +| --- | --- | +| 同步 Peer HTTP 委派 | 标准 `traceparent`/`tracestate` 注入和提取;调用端、候选选择、远端执行分别成 span。记录能力 ID、目标 Peer、执行结果;不改变现有“结果未知不能泛化重放”的规则 | +| HTTP → 持久 Job → 另一 Peer 执行 | 提交时保存可选、受限的 Trace Context;执行创建独立 trace 并用 Span Link 连接提交 span,同时记录 `job.id`。ContextVar 不能跨数据库和进程;旧 Job 没有上下文时独立起 trace | +| Cron / 周期回调 | 每次发生独立 span;Cron ID、发生时刻和创建的 Job ID 关联。不能把所有周期调用挤进一个永久 Trace | +| 浏览器直连 PostgREST/本地执行 | 在操作和 HTTP 客户端边界采集;Job 的提交上下文要覆盖直接写入路径。PostgREST 服务端/SQL 层能否贯通需查所用版本,客户端 span 不冒充数据库服务端 span | +| 并发工具/批处理 | 用 span 父子或 links 与 ToolCall ID 表达依赖;不按日志到达顺序推断执行顺序,不用跨机时间戳强行排序因果 | + +任务等待时间由已有持久时间字段与数据库时钟解释,执行耗时使用本地单调时钟。大量不同 job/thread/block/trace ID 只进入日志和 Trace 属性,不进入无界指标标签。能力、路由模板、状态、受控模型集合可用于聚合;模型名称也要控制动态值数量。[Prometheus 对标签基数的说明](https://prometheus.io/docs/practices/instrumentation/) + +Job 的可选提交上下文涉及共享数据库契约,是必要的待提议 Hub 增量。旧 Python/TS 模型能忽略额外列,但 core-py 的 exact-head readiness 仍须独立处理;模型可读不等于跨版本运行获准。不得把通用遥测元数据塞入任意 Job 的业务 parameters/state 来绕过契约,也不得将日志写入成功作为 Job claim/close 的条件。两列、容量、部署配置键与协调升级顺序已在共享契约定稿;真实数据库已证明旧消费者默认 NULL 且 claim/close 保留 carrier,模拟新 head 也证实旧 readiness 拒绝。正式 migration 与真实混合 Peer 验收仍未执行。 + +## AI O11y 与证据追溯 + +基础 Trace 应呈现 `Agent turn → model call → tool call → retrieval/resolver/Peer call → result`。直接模型调用、embedding、Organization 多模态解释也必须经过共同 AI 执行边界,不能只覆盖 Agent Query。复用已有 thread、turn、call、ToolCall、Job 和实体引用。 + +| 层次 | 记录内容 | 能承诺什么 | +| --- | --- | --- | +| 运行诊断 | provider/model、操作、耗时、流式首块延迟(适用时)、结束原因、错误类型、真实 usage、工具调用关系 | 在已采集并保留的范围内解释运行性能和失败 | +| 结果来源关联 | 检索候选、实际读取的实体、最终引用或写入结果分开记录;关联 prompt/config 版本或指纹 | 看见输入、步骤与结果的关系;记录的关联不能证明模型确实依据某段内容推理,更不能直接证明答案正确 | +| 长期证据或回放材料 | 当时实际使用的内容快照、版本、必要参数、工具结果和结果关联;有单独保留与删除契约 | 可审阅历史输入;外部模型、工具与世界状态变化意味着不能保证确定性重放 | + +第一层使用 OTel GenAI 语义约定;固定独立仓库修订 `e07f4ebacb08f56db8c4c882d117720333fbca04`,集中映射并重验升级,不跟随 Development 字段漂移。[固定 GenAI 规范](https://github.com/open-telemetry/semantic-conventions-genai/blob/e07f4ebacb08f56db8c4c882d117720333fbca04/docs/gen-ai/README.md)。模型输入输出等内容为单独 opt-in;基础性能元数据不依赖开启原始内容采集。[GenAI span 规范](https://github.com/open-telemetry/semantic-conventions-genai/blob/e07f4ebacb08f56db8c4c882d117720333fbca04/docs/gen-ai/gen-ai-spans.md) + +用量在 adapter 收到 provider response/chunk 时保留,覆盖 embedding 和流式末尾 usage;缺失是 unknown,而不是零。区分缓存和其它计费维度,以供应商返回为准。成本由用量、明确的价格版本与币种估算,实际账单仍由 provider 结算;采样 Trace 的费用和不是全量账本,应用可在采样前累计用量指标,但遥测丢失仍会影响完整性。无需仅为采集 usage 就重写所有调用方的业务结果类型;先评估 adapter 采集是否已满足查询需求。 + +源端标准 usage 字段缺失时省略,不向 token counter/histogram 写入 `-1`。目标后端可以使用 sentinel 映射,但查询、导出和面板必须先还原 unknown,不能直接求和。首选方案是允许最少的逐字段来源元数据:input/output 为 provider 或 unavailable,cost 为 estimate 或 unavailable;成本估算继续附价格版本和币种。这不重复数值,也不另建业务 authority。部分 usage 缺失时完整 total 仍未知;仅对已知部分求和并展示缺失数量。内置 AI 面板若把补零/部分和当成完整值,就使用版本化的来源感知查询;不能给误导面板贴上“可信成本”标签。 + +长期 traceability 不能依赖会采样和过期的 Trace。若要求每个答案或 AI 图谱变更以后都能解释,应由相应业务结果 owner 持久化最小证据,遥测只关联它;优先扩展已有结果/引用契约,而不是建立通用 Agent 执行库。只存 Block ID 或行更新时间不足以恢复当时的外部 Storage 字节;hash 也只能比较,不能还原。需要历史内容时必须保存当时的受控快照并承认其存储成本。 + +首个试点仅对受控合成/经授权的元数据路径配置全采样,才能检查完整性。扩大后再根据量采用采样;头部丢弃的 Trace 无法靠尾部采样补回,长时间 Job 的 Span Links 也不意味着后端会一起保留。错误/慢请求的保留规则需有实际容量验证。产品证据不受通用 Trace 采样决定。 + +专项 AI UI 作为后续消费端按需求选择,不成为采集 SDK 或业务结果的权威。若要 prompt 管理、评测数据集、人评等工作流,再验证 Langfuse 或其它工具的收益;其当前自托管包含 Web、Worker、PostgreSQL、ClickHouse、Redis/Valkey 和 Blob Storage,不应因名称是“AI tracing”就默认当作轻量附属件。[Langfuse 自托管架构](https://langfuse.com/self-hosting) + +## 数据边界与分析入口 + +当前信任边界以共享 security model 为准。需要处理的具体风险是:外部内容或异常消息携带敏感值,被采集后发给部署以外的观测运营者;未经授权的读者再从日志/Trace 取走凭据、知识内容或完整 prompt。资产是部署内容和凭据,跨越的是部署到遥测存储及其读者的边界。这里是新设计必须限定的数据出口,并非宣称现有实现存在已证实漏洞。 + +新增 OTLP 基础模式只记录明确列出的元数据和非敏感关联 ID;默认关闭新遥测时不导出这些记录。Authorization、Cookie、provider config、签名 URL、SQL 参数、任意 request/response body 不自动采集;异常消息、属性和事件也遵循内容策略,不能只过滤 HTTP headers。来源和字段策略在 Peer 第一次发送前生效;若引入 Collector,可再做一次筛选;不使用通用正则脱敏来宣称原文不会外流。G2 在正常、异常、流式结束、SDK 警告路径直接检查首次 OTLP 出口的 canary,而不只看后端清洗后的结果。Trace Context 不提供认证,浏览器上报的 Peer/resource 属性也不作为权限证明,传播目标限定在已配置的部署边界。 + +按部署分开摄取和查询权限,采集凭据不给读取/管理权限。默认不采集浏览器 session replay。内容允许时再明确保存目的、可访问者、大小限制、截断标记、保留和删除;外部 blob 引用不包含公开长期下载链接。删除业务材料不会天然删除观测副本,这一点必须进入启用内容的契约。 + +首批分析面板围绕问题建立: + +- **部署与 Peer**:运行版本、lease/readiness、请求错误率/延迟、资源饱和、数据库连接/查询压力;全局 pending Jobs 用单一读源统计,不能把各 Peer 对同一队列的观测重复相加。 +- **业务运行**:按 Job ID 找提交、等待、执行和关联 Trace;按能力看委派、未执行与结果未知;按 Source/Sink 看成功、失败、耗时和已存在的业务进展。 +- **AI**:模型/操作的延迟、失败、Token 与估算成本,单次 turn 的并行工具和检索路径,以及结果到证据引用的跳转。 +- **观测管线**:SDK 与可选 Collector 的导出失败、队列占用和丢弃,SaaS 用量/限额、保留和查询状态;后端完全失效另有外部探活,不依赖失效组件给自己发告警。 + +常用查询与仪表盘随部署配置版本化。业务 UI 保留 Job 日志能力,可增加按 Job/Trace 的诊断跳转;不在 InKCre 里重建完整日志搜索产品。告警先覆盖可采取行动的失败、积压和观测丢失;阈值依照试点基线及运行目标确定。 + +## 实现归属与实施输入 + +独立区分四个 owner:Hub 拥有跨 Unit 语义;各 Peer/AI/结果 Unit 拥有实现;部署 owner 拥有 SaaS 项目、访问和数据生命周期;W3C、OTel 与供应商拥有协议和外部服务机制。core-py 的 `docs/40-deployment/observability.md` 拟拥有部署接入与运维步骤,各 Peer 交付自己的 SDK 生命周期和连接配置,不新增专门平台仓库。若浏览器确需转发,先验证供应商能力与现有平台缺口,再由实际运行入口 owner 交付;不能因为 core-py 分发文档就把所有观测流量汇聚到 core-py。 + +| 责任表面 | 实施时的具体输入 | +| --- | --- | +| core-py `libs/obsrv`、bootstrap/middleware | 本地 opt-in 门控、标准 SDK 生命周期、专用事件 logger 与受控 span;PG 日志保持,关闭有界、后端不可达仍 ready | +| Python Job/Cron、Peer 能力调用 | 共用提交事务写 carrier,claim 成功后 Link;调用选择/执行状态不影响既有重放规则 | +| client-web 共用 core 包的 Job/Peer/DBAPIClient | SDK context 显式传递到异步边界;浏览器实际 context/OTLP/CORS 验收;Web app 的 Job 页接历史与新诊断跳转 | +| AI dialects、Agent thread/tool、Organization/embedding | provider 回包处 usage 与结束原因;区分候选、实际读取、最终引用/写入;不把 debug 原文桥接为基础日志 | +| 主机/PostgreSQL/PostgREST | 用现成接收器、exporter 或受控平台出口采集进程、连接池、存储与服务指标;SQL/任意平台日志不默认进入基础模式 | +| 独立客户端、官方 Registry | 按各自运行环境接入,覆盖状态和具体入口由工作地图逐项管理,不能把目录或共用技术栈当成已覆盖 | + +Resource 最小集合是 service 名称/版本、运行实例、deployment、已知 Peer、environment;不采集凭据派生标签。日志/Trace 另带 job/thread/turn/tool-call/实体 ID;指标维度限服务、受控操作/能力/路由/模型、结果和 error type,未知动态名称归并受控类别。Peer lease/pending Jobs 由单一读源统计,避免各 Peer 重复累计同一队列。按实例区分 SDK counter reset,聚合用 rate/increase;unknown usage 不添加零样本,另计有界 usage-missing 事件数。实际字段名以 SDK semconv 和共享契约投影为准,不建立第二套通用事件框架。 + +基础 AI traceability 在保留期内连接 Job、模型/工具、实际读取与已有最终引用;Agent Query 的 answer/references 继续留在原业务结果,Trace 不能冒充持久结果。若开启历史内容复核,扩展对应结果的版本化 evidence 引用:记录采集时刻、来源实体/版本或指纹、所用配置版本、受控快照引用及完整性状态;快照由既有 Storage/结果 owner 管理并随授权/保留删除,不建通用 Agent 执行库。此增强能力必须先确定 Q1 的保存与删除承诺,再制作该结果 owner 的 schema diff;当前没有承诺永久复现所有答案。 + +[工作地图](task-map.md)给出 Hub→Spoke→schema/各 Peer→G2→G3 的具体交付顺序和回退约束。后续只验证首选免费 SaaS 的必要准入,失败后才比较能满足条件的免费备选;不做无限产品比较。目标账户能力、实际预算、长期内容承诺和尚未执行的验收继续明确保留。 diff --git a/tasks/observability-foundation/design/hub-contract.patch b/tasks/observability-foundation/design/hub-contract.patch new file mode 100644 index 0000000..8e2667d --- /dev/null +++ b/tasks/observability-foundation/design/hub-contract.patch @@ -0,0 +1,107 @@ +diff --git a/20-product-tdd/observability-contract.md b/20-product-tdd/observability-contract.md +new file mode 100644 +--- /dev/null ++++ b/20-product-tdd/observability-contract.md +@@ -0,0 +1,65 @@ ++# 跨 Peer 可观测性契约 ++ ++本契约定义 InKCre 各 Peer 共同消费的观测身份、传播、持久 Job 因果关联及数据边界。它描述实现应满足的行为,不单独证明当前运行版本已具备这些能力。各 Unit 拥有采集实现,部署 owner 拥有采集出口、存储、查询与保留;SDK 版本、配置、后端限制和验收证据留在相应实现或部署文档。 ++ ++新增遥测必须由每个 Peer 的本地配置显式启用,缺省关闭;共享部署配置只描述身份和目的地,endpoint 或凭据存在不能自动开启。关闭时不初始化新增采集/导出,不新增 Trace Context 注入、提取或 Job carrier 捕获,仍保持现有日志与业务行为。开启也不自动替换、停用或桥接既有日志 writer;默认 PostgreSQL 日志及按 Job 查询继续可用。 ++ ++本地开关在进程启动或客户端连接重新初始化时生效;首版不承诺热切换。开启所需配置不足时报告遥测初始化问题,但不使业务 readiness 失败;关闭不能删除已有日志或持久 carrier。 ++ ++## 权威与身份 ++ ++一个 deployment 保持一个 owner 上下文,已准入 Peer 平等参与。Peer 可以直接操作共享数据库,也可以同步委派能力;观测路径不能强制所有业务绕经某个中心 Peer。[系统状态与权威](system-state-and-authority.md)继续拥有业务 authority。 ++ ++遥测用于解释已观察的运行,不决定 Job、Agent、图谱或结果的状态。采样、丢失、过期和接收顺序都不能变成业务成功、重试、执行恢复或证据完整性的依据。采集出口与观测后端不可用不应阻止应用就绪或改变业务结果;采集需有界,丢弃需可观察,关闭需有时限。 ++ ++部署标识、Peer 标识、运行实例与业务 ID 各有意义。部署配置 owner 提供稳定、非秘密的观测部署标识,同一部署的 Peer 与可选采集转发组件使用一致归属;不从凭据、数据库地址或观测产品的项目 ID 派生。重启与原部署恢复保持标识,独立 preview、克隆或新 owner 的部署分配新标识。配置尚不可用的记录可缺少该关联,不能伪造身份或以此阻止启动。 ++ ++共享部署配置使用 key `inkcre.observability`、schema `inkcre.observability.v1`;value 含 `deployment_id` UUID,以及可选、无凭据的 `otlp_http_endpoints` 对象与 `diagnostics_url`。出口对象按需声明 traces、logs、metrics 的完整 OTLP/HTTP URL;本地启用后缺省信号仍不导出,不要求所有信号共享一个 base URL;共享 value 不携带启用开关。部署启用步骤仅在缺失时创建身份,Peer 不各自生成;服务端私密采集凭据仅从运行配置提供,不进入客户端;客户端直连需使用适合其公开环境的受限写入能力,否则经过受控转发。凭据不进入该非秘密配置。各 Unit 注册和读取同一 schema;关闭的 Peer 不为新增遥测读取配置或初始化身份,开启时配置尚不可用只暂停新增遥测。 ++ ++Peer 复用现有身份,运行实例区分进程或客户端运行期,Job、Thread、ToolCall 与实体保留业务 ID。Trace/Span ID 由标准 SDK 管理,不能用业务 ID 替代。可变业务 ID 适合日志与 Trace 关联,不进入无界指标标签。资源属性用于筛选和诊断,不提供认证或权限证明。 ++ ++## 同步传播 ++ ++开启的 Peer 在同步调用中使用标准 W3C `traceparent`/`tracestate`,由所选 OpenTelemetry propagator 负责解析、版本兼容与注入。应用拥有传播目标与 carrier 容量边界,不自行维护 W3C 解析器。自动与手动采集的注入责任应唯一,实际调用使用当时的上下文。 ++ ++缺失、语义无效或未采样的上下文不改变认证、业务响应、能力选择或执行结果。Trace 解析失败不成为普通 HTTP 业务请求的拒绝理由;不打印原始非法 carrier 作为诊断。同步委派继续遵守[Peer 能力契约](semantic-retrieval-and-peer-capabilities.md),尤其不能因遥测失败重放结果未知的调用。 ++ ++传播目标限定在配置的部署边界。对外部模型、Source 或其它服务,默认观察本地客户端调用,不自动携带内部业务身份、上下文或任意 baggage。浏览器、原生 Peer 与进程内 Extension 都遵守此规则;客户端 span 不冒充尚未采集的 PostgREST 或数据库服务端执行证据。 ++ ++## 持久 Job 的因果关联 ++ ++Job 的提交和执行可以属于不同 Peer、进程与时间段。公共持久协议提供可选、受限的提交 Trace Context,由开启的创建方在创建 Job 的同一事务中保存;关闭的创建方不捕获该信息并保持缺省 NULL。它属于提交事实,领取和关闭不能将其改写成执行上下文。公共字段为 `submission_traceparent`、`submission_tracestate`,均为可选字符串、缺省 NULL、各最多 512 UTF-8 bytes;数据库协议 owner 交付约束、迁移与各语言投影,不混入 handler 的业务参数或可变执行状态。 ++ ++512 bytes 是应用持久化容量,不是 W3C tracestate 的全局最大值。创建方保存 SDK 注入的 carrier,超限 optional tracestate 在写入前整体省略并记录有界原因,保留有效 traceparent;不截断成员、不打印原值、不因观测状态重试业务写入。直接输入仍遵守普通类型和容量约束。 ++ ++Python、浏览器直写及其它直接数据库生产者采用同一提交语义。HTTP 创建入口从当前标准上下文捕获,不要求旧创建 body 接受新遥测字段。Cron 在实际创建一次 Job 时捕获该次发生的上下文,不永久继承最初创建 Cron 的请求。 ++ ++开启且成功领取的执行器建立独立执行 trace,用 Span Link 连接有效的提交上下文,并在诊断记录中保留同一 Job ID。领取失败不记成实际执行。没有 carrier 的旧 Job 仍可执行;通过普通结构约束但 W3C 语义无效的 carrier 只失去因果 link,不使持久 Job 挂起。关闭的执行器不产出新增 span,但领取和关闭必须保留行中已有 carrier。不同 Peer 的开关可以不同,因而允许提交或执行 Trace 缺失,不影响业务执行。查询可始终按 Job ID 关联,不以提交 trace 仍在保留期内为前提。 ++ ++普通字段的类型、允许表面和大小边界仍有效;观测容错不意味着接受任意输入。可选字段不自动授予跨版本运行能力。生产者与执行器必须满足[Peer 数据库运行契约](peer-database-runtime-contract.md)的准入要求,交付明确支持的 runtime/schema 组合与顺序;禁止通过写失败后重放 Job 来探测兼容。Job 的领取、取消、终态、timeout 和 Cron 发生语义仍由[知识能力契约](knowledge-capability-contract.md)拥有。 ++ ++现有 PostgreSQL 日志及其按 Job 查询能力持续保留;新增后端是按需启用的诊断增强,开启/关闭均不自动停旧 writer、搬迁历史、删除表或改变原日志关联 ID。新查询通过不构成退役旧日志的授权。 ++ ++## AI 诊断与结果证据 ++ ++模型、embedding、Agent turn、工具调用、检索、Peer 调用和最终结果通过已有业务身份关联。检索候选、实际读取与最终引用或图谱写入是不同事实;某输入与输出存在关联,不足以证明模型依据了该内容或答案正确。 ++ ++Usage 以 provider 实际返回为依据,覆盖流式结束阶段。未提供、明确为零和非零必须保持可区分;后端估算或补值不能冒充观测事实;允许用最少的字段来源元数据保持区别,缺失标记不作为真实用量参与聚合。成本估算附有价格版本与币种,并与 provider 账单区分;采样 Trace 的费用总和不是完整账本。GenAI 语义约定在实现中固定采用的修订,升级时验证消费兼容。 ++ ++短期 Trace 可采样、丢失或过期。需要长期复核的证据由对应业务结果 owner 按明确的保存与删除契约持久化,遥测只关联它;不由 tracing 自行引入 Agent 执行库、永久内容副本或确定性回放承诺。 ++ ++## 内容与访问边界 ++ ++[共享安全模型](security-boundary-model.md)定义 actor 与权限。这里需要限定的路径是:外部输入或异常携带内容、凭据,经采集写入观测存储,再被超出预期范围的读者或外部运营者取得。Trace Context、Peer/resource 属性及 CORS 均不构成授权。 ++ ++新增 OTLP 基础模式只采集声明的元数据和关联 ID。Prompt、query、工具参数/结果、引用内容、任意请求响应 body、SQL 参数和异常原文不因打开 tracing 自动获得保存许可;Authorization、Cookie、provider 配置和签名 URL 也不自动采集。基础出口只接入声明来源、字段和值来源的结构化记录,策略覆盖 span 名称、status、events、Resource 与 log body,不能仅过滤 headers 或在远端补做清洗。应用、SDK 和第三方的任意原文日志不得自动桥接;既有 PG/其它已配置 writer 按原配置保持内容行为,不要求为了保留它们再次启用新内容模式。“原文默认关闭”只约束新增 OTLP 出口,不代表整个部署无内容副本。 ++ ++内容启用独立于基础性能元数据,并明确用途、可访问者、大小与截断、保留和删除。已有 debug 开关不能自动授权向新增观测出口发送内容。删除业务源数据不会天然删除遥测副本,启用内容时必须处理这个差别。 ++ ++部署 owner 控制摄取与查询入口;采集权限不授予读取或管理权限,浏览器不持有后端管理凭据。对外托管或跨 owner 集中诊断需要明确新增数据边界,不能从非秘密部署标识推导租户隔离。 ++ ++## 后端可替换性与验证 ++ ++应用采集依赖 OpenTelemetry、OTLP 和公共业务语义,不依赖供应商 SDK、项目或 datasource ID,持久业务 schema/carrier 不保存供应商身份;后端替换应发生在出口配置、存储与消费侧,不要求改写业务采集逻辑。OTLP 兼容不保证历史数据、查询和仪表盘可以无成本迁移。部署交付记录属性映射、必要因果字段、保留或损失的协议信息、导出能力及查询迁移方法;不可恢复的字段损失不能通过默认值伪装成原始观测。 ++ ++准入应从独立后端查询核对关键字段的可消费语义、缺失与真零、必要因果关联及指标聚合,声明类型投影和损失,并覆盖持久化后的消费。发送成功不算持久化证明。故障验收还需比较真实业务结果,检查应用及可选 Collector 的队列、关闭、丢弃与恢复;只停后端、只测 SDK 或只读取数据库模型不能替代端到端验收。 +diff --git a/docs/index.md b/docs/index.md +--- a/docs/index.md ++++ b/docs/index.md +@@ -7,6 +7,7 @@ + - [Security boundary model](../20-product-tdd/security-boundary-model.md) + - [System state and authority](../20-product-tdd/system-state-and-authority.md) + - [Cross-unit contracts](../20-product-tdd/cross-unit-contracts.md) ++- [跨 Peer 可观测性契约](../20-product-tdd/observability-contract.md) + - [Knowledge capability contract](../20-product-tdd/knowledge-capability-contract.md) + - [Semantic retrieval and Peer capabilities](../20-product-tdd/semantic-retrieval-and-peer-capabilities.md) + - [Feature retrieval and media interpretation](../20-product-tdd/feature-retrieval-and-media-interpretation.md) +diff --git a/20-product-tdd/cross-unit-contracts.md b/20-product-tdd/cross-unit-contracts.md +--- a/20-product-tdd/cross-unit-contracts.md ++++ b/20-product-tdd/cross-unit-contracts.md +@@ -16,6 +16,10 @@ + - The contract owns database protocol admission, peer principals, lifecycle semantics, + readiness, JWT claims, and portable acceptance. Unit repositories own only their + implementation and provider-specific deployment mechanics. ++ ++## 跨 Peer 可观测性契约 ++ ++[跨 Peer 可观测性契约](observability-contract.md)拥有部署/Peer 观测身份、标准传播、持久 Job 提交关联、AI 诊断与内容边界。遥测不接管业务状态或协议准入;各 Unit 与部署 owner 分别交付采集、存储和查询实现。 + + ## Extension State Contract + +diff --git a/20-product-tdd/knowledge-capability-contract.md b/20-product-tdd/knowledge-capability-contract.md +--- a/20-product-tdd/knowledge-capability-contract.md ++++ b/20-product-tdd/knowledge-capability-contract.md +@@ -178,6 +178,8 @@ + running work stays running until its executor has exited and released its resources. Repeated + requests do not rewrite terminal outcomes. Stopping never promises rollback, retry, or reversal + of already dispatched external work. Each executor observes stop intent for its own active work. ++- 跨 Peer 的 Job 提交上下文与执行诊断遵守[跨 Peer 可观测性契约](observability-contract.md)。 ++ 观测关联不改变领取、执行、取消、终态或协议准入,也不提供重试与完整性语义。 + - An observer's wait budget is separate from the Job execution budget. Ending observation does not + request cancellation. A final observed record is evidence of that observation, not a claim that + the database has remained unchanged since it was read. diff --git a/tasks/observability-foundation/design/hub-review.md b/tasks/observability-foundation/design/hub-review.md new file mode 100644 index 0000000..e23d91e --- /dev/null +++ b/tasks/observability-foundation/design/hub-review.md @@ -0,0 +1,20 @@ +# Hub 契约差异评审 + +[hub-contract.patch](hub-contract.patch) 保留基于旧 Hub `42f7bad1c61e57b5e0ebf55e27815ddc2ae913fa` 的设计证据。授权实施后已重新基于 `03d0c54` 检查并应用到 Hub 源,提交 `a603b036fd41dae5a471dea24cd7609bb4927bd5`,发布 [draft PR #33](https://github.com/InKCre/docs/pull/33);两 Spoke 各自独立更新引用。下文的未应用/未授权描述属于此前设计阶段,不再表示当前状态;当前控制见 [packet](../packet.md)。 + +| 变更对象 | From → To | 影响与边界 | +| --- | --- | --- | +| `20-product-tdd/observability-contract.md` | 无独立观测契约 → 默认关闭/现有日志保留、身份、传播、Job 因果、AI unknown、内容和后端可替换性要求 | Python、TS 及其它 Peer 的共同可观察行为;不承诺当前实现完成 | +| `docs/index.md` | 既有共享导航 → 增加新契约链接 | 只改人工维护区域,生成 SVC 区域保持原样 | +| `20-product-tdd/cross-unit-contracts.md` | 既有合同索引 → 明确观测语义 owner | 引用新文档,不复制整份契约 | +| `20-product-tdd/knowledge-capability-contract.md` | Job 生命周期约束 → 增加观测关联的连接 | 不修改领取、终态、取消、重试或 Cron 语义 | + +归位独立判断如下:Hub 拥有跨 Unit 行为;数据库协议 owner 交付持久字段和准入;各 Peer 实现本地 opt-in、SDK/context 与 provider 边界;部署 owner 配置托管采集、存储和查询,按实际缺口使用可选转发组件。W3C 与 OTel 的协议机制由标准依赖拥有,InKCre 只补业务关联和边界。按 D8 改为 SaaS 优先,core-py 拥有相关部署接入文档;具体供应商、配方和位置不进入 Hub 契约。 D10 已获用户同意选用 Grafana Cloud Free,但公共契约仍不绑定它;关闭态保持原日志,开启也不自动停用 PG。 + +新增 `-1`/来源标记实验与 SaaS 文档判别见[本轮报告](../experiments/sentinel-saas-20261003.md),目标 Cloud 尚未验收。既有支持证据是标准 Python/JS 互操作、实际旧 Python/TS 数据库消费者与 readiness、carrier 容量、OpenObserve 的 AI 缺失值反例、Tempo 重启前后的因果/属性保留,以及五组件 Grafana 诊断闭环。它们见[实验记录](../experiments/README.md)。patch 要求的真实业务故障隔离、浏览器传播、内容出口、正式 migration/完整启动和历史导出仍需实现验收;不能用文档发布替代这些证据。 + +D10 增加本地显式 opt-in,关闭时不捕获新 carrier,更新已有 Job 仍保留字段;撤销 PG writer 退役和保留旧日志需再次授权内容模式的约束。carrier 公共字段/容量保持;未发布配置由单一 base URL 改为按信号完整 URL,允许客户端专用受限写入与受控转发,均已进入 patch,避免跨语言各自发明。SQL DDL、配置初始化命令、生成类型、SDK 生命周期和协调升级步骤由[实现输入](shared-contract.md)细化,不进入公共行为文档。当前后端的版本、资源和 Link flags 损失只在部署实验中记录,不把某版本存储缺口固化成共享标准。内容长期保存和生产预算仍等待实际需求,不从本差异推导永久证据或生产部署授权。 + +2026-10-03 已对原 Hub 工作树运行 `git apply --check`,并在临时目录应用后核对新增链接、四个文件范围及生成 SVC 区域,均通过;未把 patch 应用到源仓。正式落地前重查基线与本地差异。交付依次是 Hub 源、经明确指令提交/推送、Spoke 引用、各 Spoke 实现,分别保持独立变更边界。 + +仓库 AGENTS 明确要求源码修改批准,并要求提交仅依明确 Human 指令。当前已授权的任务包与隔离实验足以生成和验证此差异;本文件不自行扩大为修改应用、提交、推送或生产发布的授权。 diff --git a/tasks/observability-foundation/design/shared-contract.md b/tasks/observability-foundation/design/shared-contract.md new file mode 100644 index 0000000..b0cacb1 --- /dev/null +++ b/tasks/observability-foundation/design/shared-contract.md @@ -0,0 +1,75 @@ +# 跨 Peer 观测契约与持久形状 + +状态:2026-10-03 契约候选已按显式 opt-in、PG 保留与 SaaS 方向修订,尚未发布。本文细化[整体设计](../design.md)的跨语言实现输入;[Hub patch](hub-contract.patch)承载公共语义,数据库 schema、语言投影和迁移由 core-py 数据库协议 owner 交付。实验结果见[数据库与三信号报告](../experiments/convergence-20261003.md)。 + +## 显式启用与现有日志 + +新增遥测通过每个 Peer 本地的 `telemetry_enabled` 开关控制,缺省 false。Python 拟映射为 `obsrv.telemetry_enabled` / `OBSRV__TELEMETRY_ENABLED`;浏览器在现有部署连接配置中提供同义布尔值,原生 Peer 使用其运行配置。该开关不是既有 `logging_backend` 或 `ENABLE_LOG_BACKEND`,不进入共享部署配置,不添加第二个部署级总开关。仅存在 endpoint、token、diagnostics_url 或安装 SDK 不能启用。 + +| 有效状态 | 新增遥测 | 已有日志与业务 | +| --- | --- | --- | +| 未配置或 false | 不初始化新增 SDK/provider/instrumentation/exporter,不发 OTLP,不新增传播与 carrier 捕获 | 保持 stdout 与原 logging_backend,默认 PostgreSQL;现有 Job 查询可用 | +| true,出口有效 | 按已配置的信号初始化标准 SDK/OTLP;没有配置的信号不导出 | PG writer 不替换、不停写;业务不依赖后端可用 | +| true,配置不足或出口初始化失败 | 报告明确且不含凭据的本地配置问题,不启用受影响信号 | 业务正常启动,PG 路径保持;不靠记录业务重试来恢复遥测 | +| 由 true 改为 false | 下一次进程启动/浏览器连接重新初始化后停止新采集;前一运行期正常关闭仍可有界排空 | 不删除 PG 历史或 Job carrier,不需要恢复/重建 PG writer | + +第一版不承诺热切换。关闭态不为新增遥测额外查询共享配置或要求创建部署标识,也不增加队列、轮询和 SDK 关闭等待。显式设置已有 logging_backend 为 none/logtail 等仍按原配置执行;代码和 .env 模板已默认 PG,Compose 未配置 fallback 为 none 的不一致列入实现对齐。 + +## 身份和配置 + +部署、Peer、运行实例和业务操作各有身份。部署标识使用一次分配的 UUID;Peer 复用已注册 UUID;`service.instance.id` 每次进程或浏览器运行期生成。Job/Thread/ToolCall/实体 ID 保留原类型和意义,Trace/Span ID 由 SDK 生成。启动期身份不全可省略,不生成假 Peer,也不阻挡业务就绪。 + +复用现有 `configs`,选择 key `inkcre.observability`、schema `inkcre.observability.v1`,value 包含必需的 `deployment_id` UUID,以及可选的 `otlp_http_endpoints` 和 `diagnostics_url`。前者是对象,按需包含 traces、logs、metrics 三个完整 OTLP/HTTP 出口 URL,本地开关开启后,缺省的信号仍不启用远端导出;不自动拼接固定后缀。后者为诊断入口基础地址,不保存供应商查询模板。所有 URL 均无凭据,不允许 userinfo 或带 token 的 query。进程端映射标准 per-signal exporter 配置;显式本地运行配置优先于共享默认值,浏览器读取同一投影与其连接设置。配置 schema 在现有 DeploymentConfigManager 注册,由各语言读取对应投影,不增加配置服务。 + +部署安装/启用步骤在已有配置缺失时一次创建身份;并发初始化以数据库唯一 key 和 insert-on-conflict 保留已存在值,不能用 replace 重置 ID。开启的 Peer 只读取;本机启动期间配置不可用则保留本地诊断、暂不远端导出,配置恢复后在本地开关仍开启的前提下重新初始化接入。Exporter 必须先满足本地 telemetry_enabled=true,再检查有效 endpoint 与出口配置;任何共享配置都不能覆盖本地 false,不影响 readyz。服务端私密 OTLP ingest token 属于运行配置,不进入客户端;供应商专为公开客户端提供的受限写入能力需单独准入,否则使用受控按请求转发。凭据不进入该非秘密 value 或诊断跳转 URL。 + +部署 owner 保证各出口使用同一 deployment ID;若存在 Collector/转发入口,其受控配置使用相同归属,不信任客户端自报属性作为权限。重启/原地恢复保留 ID;克隆、preview 和新 owner 的部署在启用出口前显式换 ID。首版不声称能自动识别数据库克隆。 + +## 同步传播 + +开启的 Peer 采用标准 OTel W3C propagator 注入/提取 `traceparent`、`tracestate`,不复制 parser。同步 Peer 请求在实际调用边界注入当前 SDK context;auto/manual 只保留一个注入 owner。调用端的选择和远端执行分别记录,继续遵守原有 not-executed 与结果未知不能泛化重放的规则。 + +缺失、语义无效、未采样的 carrier 不改变认证与业务响应。传播 allowlist 来自已经配置的部署 Peer endpoint;外部 provider/source 默认只有本地 client span,不外传内部 context 或 baggage。浏览器的 CORS 允许项不能代替认证,PostgREST client span 不能冒充 server/SQL span。 + +## Job 的公共形状 + +| 字段 | 类型和边界 | 写入与读取责任 | +| --- | --- | --- | +| `submission_traceparent` | nullable text,缺省 NULL,UTF-8 最多 512 bytes | 创建方保存 SDK 注入的值;当前采用 SDK 注入 55 bytes 的 version 00 | +| `submission_tracestate` | nullable text,缺省 NULL,UTF-8 最多 512 bytes | 可选 vendor state;capture 时超限省略整个字段,保留有效 traceparent | + +512 bytes 是应用持久化政策,不是 W3C tracestate 的全局最大值。SDK 实测可产生 1109 bytes;省略 optional state 的决定必须发生在业务写事务之前,记录固定原因的丢弃计数,不输出原始值,也不为遥测增加写失败重放。SQL 只验证普通类型/容量,不校验 W3C 语法。直接数据库调用传入超限字段仍按普通协议错误拒绝,不能借容错接受无限输入。 + +开启的创建方在创建 Job 的同一事务内保存 carrier;Python REST 从当前 SDK context 捕获,不在旧 JobCreateForm 中添加字段。Python 的共用 create_in_uow、Cron materialize/run_now、TS JobManager 的 PostgREST 创建都落在此语义。Cron 每次真正创建 Job 时捕获该次发生的 context,不能永久继承最初创建 Cron 的请求。 + +开启的执行器成功 claim 后建立独立 trace,以标准 Span Link 连接有效提交 span,记录 Job ID;新执行 context 不能覆盖 submission 列。旧 Job 的 NULL 和有界语义损坏只失去 Link,不阻止执行。读取语义用 SDK;SDK malformed tracestate WARNING 只留在本地诊断,不桥接到基础远端 logger。查找路径始终允许 Job ID,不要求提交 Trace 尚在保留期内。 + +关闭的创建方不捕获新增 carrier,两个字段按缺省 NULL 写入;关闭的执行方不创建 span,但 claim/close 必须保留行中已有 carrier。开→关、关→开与关→关组合都保持现有 Job 结果语义;开→关可能只有提交 Trace,关→开可能只有独立执行 Trace,不能伪造完整链路。这里描述同一获准 schema 上的配置混用,不授予旧二进制跨 migration head 混跑能力。 + +## 兼容、准入与升级 + +| 组合 | 已观察行为 | 交付约束 | +| --- | --- | --- | +| 旧生产者 → 新 nullable 列 | 实际 Python repository 与 TS/PostgREST 创建默认 NULL | 仅在 runtime 已获协议准入时可运行 | +| 新 carrier 行 → 旧模型/执行方法 | 旧 Python/TS 模型忽略新列;实际 claim/close 保留两列 | 不由该证据推导旧进程在新 migration head 获准运行 | +| 有界语义损坏 / 容量超限 | 前者可存,后者被实际数据库约束拒绝;SDK 解析实验已完成 | 新执行器的真实业务回归在 G2 验收 | +| 新 migration head → 旧 readiness | 实际检查对模拟新 head 拒绝 | 不改变 exact-head,不承诺滚动混跑 | +| 新生产者 → 旧 schema | 无受支持写入路径 | 遵循 admission;不捕获失败后删字段重试 Job | + +首版按协调升级交付:先发布 Hub 契约与 Spoke 引用,再发布 schema/生成协议/Python/TS 匹配版本;部署时暂停新提交、停用领取并排空或按既有机制取消执行,备份后迁移、刷新 PostgREST schema cache、更新所有参与的运行版本、检查 readiness 再恢复。旧浏览器页/离线客户端重连需刷新到匹配版本。任务不增设零停机框架;若有此要求,重新决策发布模型。 + +将本地遥测开关改为 false 并重启进程或重新初始化客户端连接,是观测故障的首选回退,不涉及 schema 回退。旧二进制因 exact-head 不可直接替换新版本;数据库 downgrade 只允许在维护窗口排空、备份并验证的独立操作,不能为回退可选遥测删除业务 Job。正式 migration、完整启动和协调升级演练仍是 G2 验收,不把实验 DDL 当发布物。 + +## PostgreSQL 日志保留与 AI 关联 + +`job.` 及现有 PG `trace_id`/`span_id` 保持原语义,不改作标准 OTel ID。Job 页的当前与历史日志始终从既有路径读取;已配置的新诊断入口是附加能力,新遥测事件带独立 Job/trace/span 属性。不以新查询验收通过作为停用 PG writer、删除表或搬迁历史数据的条件,本任务不再安排旧 writer 退役。 + +新增 OTLP 只接专门的结构化事件 logger,不桥接 PG 历史、任意应用日志或 agent_debug 原文。现有 PG/Logtail 内容行为继续由原配置控制,不要求为了保持当前行为开启新的内容模式;“原文默认关闭”只描述新增 OTLP 出口,不能泛指整个部署。 + +AI 的 metadata、usage unknown、结果 authority 和内容边界遵循整体设计。`AgentQueryResult.answer/references` 及现有结果保存仍由业务 owner 负责;基础阶段从 Job/实体 ID 关联诊断,不能将 Trace 留存冒充答案或图谱的历史快照。未来有长期快照承诺时,先扩展相应结果/引用契约,再启用内容存储。 + +## 变更和验证边界 + +本提案从“无标准提交上下文/统一部署关联”变为上述两个 nullable 字段与一个共享配置。影响数据库协议、生成 TS 类型、各语言采集初始化和 Job 查询入口;不改变 Job claim、取消、终态、重试、业务 authority 或认证模型。确切规范键必须随 Hub 合同一并评审,SQL、SDK wiring、后端版本和运行参数分别归实现/部署文档。 + +标准 SDK、实际旧数据库消费者、容量边界与合成三信号已提供 G1 支撑。V0—V6 的默认行为、真实跨 Peer、浏览器、错误路径和出口验证仍需执行;Tempo 后端历史 Link flags 损失不能放宽实时传播/持久 carrier,也不能从历史默认值推断原始采样状态。 diff --git a/tasks/observability-foundation/experiments/.gitignore b/tasks/observability-foundation/experiments/.gitignore new file mode 100644 index 0000000..65b83b1 --- /dev/null +++ b/tasks/observability-foundation/experiments/.gitignore @@ -0,0 +1,2 @@ +node_modules/ +runtime/ diff --git a/tasks/observability-foundation/experiments/README.md b/tasks/observability-foundation/experiments/README.md new file mode 100644 index 0000000..ff371ce --- /dev/null +++ b/tasks/observability-foundation/experiments/README.md @@ -0,0 +1,87 @@ +# G1 前轮传播与后端判别 + +> 后续修订:Sir 明确 serverless/scale-to-0 后,默认自建选择已由 [D8](../decisions.md) 替代;当前方向与缺失值判别见 [SaaS 修订报告](sentinel-saas-20261003.md)。本文件保留当时的实验事实和决策背景。 + +2026-10-01—03 使用合成数据检查标准传播、旧模型读取及后端语义。结论是保留 OTel/OTLP 与部署自有 Collector,结束 OpenObserve 1.0.4 的统一后端候选验证,继续 Tempo 3.1.0 的 Trace 路线。本轮尚未包含数据库与五组件闭环,后续结果见[2026-10-03 收敛报告](convergence-20261003.md)。本文保留当时的实验事实;选择与重开条件由[任务决定](../decisions.md)拥有。没有运行 InKCre 业务旅程,不能把这些结果算成 G2 通过。 + +## 判别结果 + +| 检查 | 实际观察 | 对下一步的影响 | +| --- | --- | --- | +| Python ↔ JavaScript 标准传播 | sampled/unsampled、未来版本、缺失/损坏/零 ID、tracestate 及独立 Job trace/Link 均按 SDK 结果核对;Python asyncio 和 JS 显式 context 并发隔离通过 | 复用官方 propagator;JS 在 Node 运行,浏览器异步 context 尚未验证 | +| SDK 诊断日志 | Python `opentelemetry.trace.span` 的 malformed tracestate WARNING 含合成原文 canary | 实现必须在出口前限制该诊断;不因不记录 HTTP body 就宣称没有内容副本,也不重写 W3C parser | +| 旧 Job 模型 | 实际 Python SQLModel 与 TS Zod 接受缺失、NULL、额外 carrier 字段,忽略未知列;旧 REST create form 拒绝额外字段 | REST 从 HTTP 上下文捕获,不向旧 body 塞字段。模型可读不证明数据库更新或 runtime admission 兼容 | +| OpenObserve 三信号 | 独立 API 查到提交/执行/AI 三个 spans、关联日志、counter=3、histogram count=2/sum=0.4 和各桶;重启后仍可查 | 基本摄取成立;累积指标不能跨导出快照求和 | +| OpenObserve 属性/Links | Trace 的整数 Job ID 变字符串;Link 属性被展开、整数变字符串,Link ID/tracestate 可读 | 表示变换需要消费端声明,不等于原始 OTLP 无损导出 | +| OpenObserve AI 缺失值 | 未提供 usage/cost 的 AI span 被写入 usage=0、cost=-0.0;缺失 agent.version 被 service.version 填充。显式零和显式非零也可读 | unknown 与真零在摄取时已不可区分,查询层无法还原;当前版本不符合本任务 AI 语义 | +| Collector 后端中断 | 4096 spans、16 次请求全部 HTTP 200;最终 enqueue_failed=3712、send_failed=384,队列回零 | 接收成功不是持久化成功;丢失可观察。没有执行业务,V6 仍待验证 | +| Tempo 查询与重启 | 按整数 Job ID 找到四条 trace;重启前后属性的存在性、类型和值,以及 Link ID、tracestate、属性均与输入一致 | 当前 AI/因果语义的窄门槛通过;可以继续检验日志、指标与 UI 集成 | +| Tempo Link flags | 输入 flags=1,查询中缺失;固定版本存储 schema 也无该字段 | 完整 OTLP 保真不通过。不得把查询默认零解释成原始未采样,更不能用历史记录恢复传播或推断完整性 | + +旧模型检查不访问数据库。源码另证实 core-py 的 readiness 要求 migration heads 精确一致,因此“加 nullable 列”不能保证旧 core 在新 schema 上启动;旧模型检查也没有验证已经运行的旧 Peer 能否持续工作。发布必须独立处理协议准入。 + +OpenObserve 的三个 AI 判别样本分别是 absent、真实 zero、显式 usage 8/3 与 cost 0.125。固定标签源码表明 LLM enrichment 会填充缺失值;`ZO_MODEL_PRICING_ENABLED=false` 只关闭数据库价格查询,仍有内置价格回退,不构成禁用 enrichment 的证据。主 Agent 与 advisor 据此收束该候选,不增加影子字段或修复层。依据:[摄取入口](https://github.com/openobserve/openobserve/blob/v1.0.4/src/core/src/traces/mod.rs)、[usage/cost 处理](https://github.com/openobserve/openobserve/blob/v1.0.4/src/core/src/traces/otel/processor.rs)、[版本候选字段](https://github.com/openobserve/openobserve/blob/v1.0.4/src/config/src/meta/gen_ai.rs)。 + +Tempo flags 限制由[固定版本存储定义](https://github.com/grafana/tempo/blob/v3.1.0/tempodb/encoding/vparquet5/schema.go)与实际查询相互核对。当前执行采样发生在 SDK,Job carrier 是传播依据,后端查询只消费业务 ID 和因果关联,因此该历史存储损失暂不阻挡窄范围候选准入;SDK、Collector 和 Job carrier 的标准传播要求不放宽。未来消费者需要历史 flags 时重开。 + +## 版本、拓扑与资源 + +| 组件 | 实测版本与固定来源 | +| --- | --- | +| Python SDK / OTLP exporter | 1.45.0,脚本 PEP 723 隔离依赖 | +| JavaScript | Node 24.15.0;OTel API 1.9.1、core/sdk-trace-base 2.11.0;[package-lock.json](package-lock.json) | +| Collector | 0.162.0;`otel/opentelemetry-collector@sha256:310a800ad69ee430e7c541796852a242c9c7db97aaad4daa5ccf843c525fbdb2` | +| OpenObserve | 1.0.4;`openobserve/openobserve@sha256:d4a878fac1f6c56003764f7f2a1625668917388f167e222c8c810de3f54c56ba` | +| Tempo | 3.1.0;`grafana/tempo@sha256:3076b8dcdfb32fd6bc5ccef85e7b7313e6199b9cb84366257fc17ecb696db5fd` | + +实验经已有 SSH 主机 `wsl.win-ws.localhost` 使用 Windows Docker CLI 访问 Docker Desktop Linux x86_64 daemon;宿主报告 12 CPU、约 16.745 GB 内存。原生 Linux Docker CLI 指向另一 daemon,不能替换调用。两个实验依次占用同一个 OTLP 转发端口,不同时启动。 + +每套实验限制合计 2 CPU/4 GiB:后端 1.5 CPU/3.5 GiB、Collector 0.5 CPU/512 MiB。各信号出口队列按字节限制为 1 MiB,有限重试。后端使用独立本地数据卷,平台遥测关闭;Tempo 单体 `target=all`,不引入 Kafka。普通 bridge 网络与 loopback 端口发布不等于出站隔离。 + +OpenObserve 重启后小样本查询 20 次,客户端 RTT 中位 56.51 ms,按排序第 19 项计算 P95 为 63.01 ms。后端中断实验的 Collector 队列采样峰值 912,150 bytes,过程 RSS 采样约 95 MB;4096 spans 最终全部丢弃。Tempo 重启后四条 trace 下 Docker 快照为后端 40.66 MiB、Collector 25.82 MiB。它们是小样本和时间点观察,不能用于全栈容量承诺、每日增长或应用开销比较;实验没有生产负载模型。 + +Tempo 本地部署仅用于此实验。本轮未验证正式入口鉴权、Grafana/Loki/Prometheus 集成、保留、删除、备份恢复与容量;Tempo idle scheduler 出现重复 `no jobs found` 错误日志,运维噪声待复核。重启保留样本不等于备份恢复通过。[官方单体本地部署说明](https://grafana.com/docs/tempo/latest/set-up-for-tracing/setup-tempo/deploy/locally/linux/) + +## 证据入口 + +- [标准传播结果](evidence/openobserve-20261001/propagation.json)、[三信号输入标识](evidence/openobserve-20261001/expected.json)、[独立查询和延迟](evidence/openobserve-20261001/readback.json)。 +- [AI 判别输入、后端输出与 Link](evidence/openobserve-20261001/ai-projection.json)、[中断原始计数和资源采样](evidence/openobserve-20261001/failure.json)。 +- [Tempo 输入](evidence/tempo-20261003/tempo-input.json)、[重启前](evidence/tempo-20261003/tempo-before.json)、[重启后](evidence/tempo-20261003/tempo-after.json)、[Job 查询前](evidence/tempo-20261003/tempo-search-before.json)、[Job 查询后](evidence/tempo-20261003/tempo-search-after.json)。 +- [Tempo 落盘指标](evidence/tempo-20261003/tempo-metrics.txt)、[重启后资源快照](evidence/tempo-20261003/tempo-stats.jsonl)。重启前 `blocks_completed_total=1`、`local_blocks_flushed_total=1`,用于区别纯内存读回。 + +第一次 Tempo 发送被中断,没有接收成功证据;10 月 3 日先核对空查询与 ready,再发送并完成读回/重启验证。不得把中断的首次尝试记成已接收数据丢失。归档不包含凭据,随机实验凭据仅在忽略的 `runtime/credential.json`,权限 0600。 + +## 复现与资源归属 + +以下命令从 core-py 根运行。脚本是本任务的判别实验,不是可移植生产工具。先在实验目录执行 `npm ci --ignore-scripts --no-audit --no-fund`;旧 TS 模型检查还依赖相邻 client-web 的现有 esbuild 0.28.1 路径。`propagation.py` 需已有 `runtime/` 目录,可用 `mkdir -p tasks/observability-foundation/experiments/runtime` 创建。 + +```bash +pdm run tasks/observability-foundation/experiments/propagation.py +pdm run python tasks/observability-foundation/experiments/legacy-consumers.py +node tasks/observability-foundation/experiments/legacy-consumers.mjs +python3 tasks/observability-foundation/experiments/lab.py up +pdm run tasks/observability-foundation/experiments/signals.py +python3 tasks/observability-foundation/experiments/readback.py +python3 tasks/observability-foundation/experiments/ai-projection.py +python3 tasks/observability-foundation/experiments/failure.py +python3 tasks/observability-foundation/experiments/lab.py stop +python3 tasks/observability-foundation/experiments/tempo-lab.py up +python3 tasks/observability-foundation/experiments/tempo-probe.py send +python3 tasks/observability-foundation/experiments/tempo-probe.py before +python3 tasks/observability-foundation/experiments/tempo-lab.py restart +python3 tasks/observability-foundation/experiments/tempo-probe.py after +python3 tasks/observability-foundation/experiments/tempo-lab.py stop +``` + +启动后先检查服务 ready;异步入库需要等待可查询,不能连续执行上面所有命令后将短暂不可见当成不保留。Tempo 重启实验应先从 `/metrics` 确认样本完成落盘,再重启并等 `/ready` 返回 200。查询脚本按近一小时搜索,历史复核使用归档时间范围。`ai-projection.py` 会报告明确的语义失败布尔值,退出码零不表示候选通过。`failure.py` 在 finally 中重启后端。 + +WSL interop 曾超时;当时经只读核对使用命令级 `LAB_WSL_INTEROP=/run/WSL/1905_interop` 恢复 Windows CLI。socket 会变化,不把该值当成永久配置,不修改宿主环境来迁就脚本。`internal:true` 网络首次未发布端口,现用普通 bridge;失败启动不作为摄取成功证据。 + +2026-10-03 已独立检查下列两个项目的四个容器均为 `exited`,两个 SSH control socket 均不存在。卷保留合成数据,随活跃父任务保留: + +| Compose project | 保留卷 | +| --- | --- | +| `inkcre-o11y-g1-b0a97f7c` | `inkcre-o11y-g1-b0a97f7c_data` | +| `inkcre-o11y-tempo-g1-b0a97f7c` | `inkcre-o11y-tempo-g1-b0a97f7c_data` | + +各配方 `stop` 停止并关闭隧道,`remove` 仅删除对应实验项目及卷。结束父任务时决定保留或删除;不要调用 `svc dev stop database` 清理本实验。未修改 SVC 数据库资源、应用源码、Hub 源、共享引用或仓库依赖。 diff --git a/tasks/observability-foundation/experiments/ai-projection.py b/tasks/observability-foundation/experiments/ai-projection.py new file mode 100644 index 0000000..07ce997 --- /dev/null +++ b/tasks/observability-foundation/experiments/ai-projection.py @@ -0,0 +1,168 @@ +"""Discriminate absent/zero/explicit source facts and linked-span export retention.""" + +import json +from pathlib import Path +import time +import urllib.request +import uuid + +from readback import query + + +def attrs(values): + return [ + { + "key": key, + "value": {"intValue": str(value)} + if type(value) is int + else {"doubleValue": value} + if type(value) is float + else {"stringValue": value}, + } + for key, value in values.items() + ] + + +def main(): + run_id = uuid.uuid4().hex + parent_trace, parent_span = uuid.uuid4().hex, uuid.uuid4().hex[:16] + now = time.time_ns() + cases = { + "absent": {}, + "explicit": { + "gen_ai.agent.version": "agent-7", + "gen_ai.usage.input_tokens": 8, + "gen_ai.usage.output_tokens": 3, + "gen_ai.usage.cost": 0.125, + }, + "zero": { + "gen_ai.usage.input_tokens": 0, + "gen_ai.usage.output_tokens": 0, + "gen_ai.usage.cost": 0.0, + }, + } + spans = [ + { + "name": "submission", + "traceId": parent_trace, + "spanId": parent_span, + "startTimeUnixNano": str(now), + "endTimeUnixNano": str(now + 1000), + "attributes": attrs({"inkcre.lab.run_id": run_id}), + } + ] + for name, fields in cases.items(): + spans.append( + { + "name": name, + "traceId": uuid.uuid4().hex, + "spanId": uuid.uuid4().hex[:16], + "startTimeUnixNano": str(now), + "endTimeUnixNano": str(now + 1000), + "attributes": attrs( + { + "inkcre.lab.run_id": run_id, + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": "synthetic", + **fields, + } + ), + "links": [ + { + "traceId": parent_trace, + "spanId": parent_span, + "traceState": "inkcre=synthetic", + "flags": 1, + "attributes": attrs( + {"inkcre.link.reason": "job-submission", "inkcre.link.sequence": 7} + ), + } + ], + } + ) + payload = { + "resourceSpans": [ + { + "resource": { + "attributes": attrs( + {"service.name": "inkcre-projection-probe", "service.version": "service-3"} + ) + }, + "scopeSpans": [{"spans": spans}], + } + ] + } + request = urllib.request.Request( + "http://127.0.0.1:34318/v1/traces", + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=5) as response: + assert response.status == 200 + time.sleep(3) + result = query(f"SELECT * FROM default WHERE inkcre_lab_run_id = '{run_id}'", "traces") + (Path(__file__).parent / "runtime/ai-projection.json").write_text( + json.dumps({"input": payload, "output": result}, indent=2) + "\n" + ) + rows = {row["operation_name"]: row for row in result["hits"]} + assert set(rows) == {*cases, "submission"} + explicit = rows["explicit"] + assert explicit["gen_ai_agent_version"] == "agent-7" + assert explicit["gen_ai_usage_cost"] == 0.125 + for key, value in { + "gen_ai_usage_input_tokens": 8, + "gen_ai_usage_output_tokens": 3, + }.items(): + assert explicit[key] == value + links = {name: json.loads(rows[name]["links"])[0] for name in cases} + for link in links.values(): + assert link["context"]["traceId"] == parent_trace + assert link["context"]["spanId"] == parent_span + assert link["context"]["traceState"] == "inkcre=synthetic" + # OpenObserve's SpanLink projection flattens attributes into the link object. + assert link["inkcre.link.reason"] == "job-submission" + assert link["inkcre.link.sequence"] == "7" # Numeric custom attributes are stringified. + parent = query( + f"SELECT * FROM default WHERE trace_id = '{parent_trace}' AND span_id = '{parent_span}'", # noqa: E501 + "traces", + ) + assert parent["hits"][0]["operation_name"] == "submission" + (Path(__file__).parent / "runtime/ai-projection.json").write_text( + json.dumps({"input": payload, "output": result, "linked_submission": parent}, indent=2) + + "\n" + ) + keys = [ + "gen_ai_agent_version", + "gen_ai_usage_input_tokens", + "gen_ai_usage_output_tokens", + "gen_ai_usage_cost", + ] + print( + json.dumps( + { + "explicit_fields_preserved": True, + "link_attributes_and_state_exported": True, + "link_attribute_type_preserved": all( + type(link["inkcre.link.sequence"]) is int for link in links.values() + ), + "missing_usage_remains_unknown": all( + key not in rows["absent"] + for key in [ + "gen_ai_usage_input_tokens", + "gen_ai_usage_output_tokens", + "gen_ai_usage_cost", + ] + ), + "missing_agent_version_remains_absent": "gen_ai_agent_version" + not in rows["absent"], + "absence_vs_zero": { + name: {key: rows[name].get(key) for key in keys} for name in cases + }, + }, + indent=2, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/tasks/observability-foundation/experiments/ai_otlp_probe.py b/tasks/observability-foundation/experiments/ai_otlp_probe.py new file mode 100644 index 0000000..5c35820 --- /dev/null +++ b/tasks/observability-foundation/experiments/ai_otlp_probe.py @@ -0,0 +1,557 @@ +"""Synthetic HTTP provider -> real OpenAI SDK -> first OTLP outlet acceptance. + +Run from the repository using pdm run python and this script path. +Database lookups use synthetic domain objects; Tools, Resolver, Thread, Sink, +provider HTTP, SDK parsing and OTLP exporters execute their production code. +""" + +import asyncio +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import json +import os +import secrets +from pathlib import Path +import sys +import threading + +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) +os.environ["INKCRE_ENV_FILE"] = "" +os.environ["DATABASE_URL"] = "postgresql://probe:probe@127.0.0.1:1/probe" +os.environ["JWT_SECRET"] = secrets.token_hex(32) + +import pydantic +from opentelemetry import trace +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( + ExportTraceServiceRequest, +) +from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ExportLogsServiceRequest +from opentelemetry.proto.collector.metrics.v1.metrics_service_pb2 import ( + ExportMetricsServiceRequest, +) + +from app.business.ai import AIManager +from app.business.agent import ( + BoundAgentTool, + InMemoryThreadPersistenceBackend, + Thread, + ThreadState, +) +from app.schemas.ai import ( + AIModelModel, + AIProviderModel, + ChatCapability, + EmbeddingCapability, + FunctionTool, + TextContentPart, + UserMessage, +) +from app.business.agent.contracts import ToolExecutionError +from app.business.agent import AgentManager +from app.business.info_base.main import InfoBaseManager, RetrievalBranch +from app.business.info_base.services import BlockService +from app.business.info_base import tools as info_tools +from app.business.info_base.resolver import tools as resolver_tools +from app.business.info_base.resolver.text import TextResolver +from app.business.sink.agent_query import AgentQuerySink +from app.schemas.info_base.block import BlockModel +from app.schemas.info_base.relation import RelationModel +from app.schemas.semantic_retrieval import ( + SemanticRetrievalResult, + BlockSemanticRetrievalMatch, + RelationSemanticRetrievalMatch, +) +from app.schemas.job import JobModel +from app.schemas.sink import SinkModel +from libs.obsrv.telemetry import start_telemetry, close_telemetry, operation + +CANARY = "CONTENT_CANARY_4f7ca94e" +exports = [] +requests = [] + + +class Server(BaseHTTPRequestHandler): + def log_message(self, *_args): + pass + + def do_POST(self): + body = self.rfile.read(int(self.headers["Content-Length"])) + if self.path.startswith("/otlp/"): + exports.append((self.path, body)) + self.send_response(200) + self.end_headers() + return + request = json.loads(body) + requests.append(request) + model = request["model"] + if model == "error": + self.send_response(400) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write( + json.dumps({"error": {"message": CANARY, "type": "invalid_request_error"}}).encode() + ) + return + self.send_response(200) + self.send_header( + "Content-Type", "text/event-stream" if request.get("stream") else "application/json" + ) + self.end_headers() + usage = { + "prompt_tokens": 7, + "completion_tokens": 0, + "total_tokens": 7, + "prompt_tokens_details": {"cached_tokens": 3}, + "completion_tokens_details": {"reasoning_tokens": 0}, + } + if request.get("stream"): + for choice, chunk_usage in [ + ([{"index": 0, "delta": {"content": CANARY}, "finish_reason": None}], usage), + ([{"index": 0, "delta": {}, "finish_reason": "stop"}], None), + ([], usage), + ]: + chunk = { + "id": CANARY, + "object": "chat.completion.chunk", + "created": 1, + "model": CANARY, + "choices": choice, + "usage": chunk_usage, + } + self.wfile.write(("data: " + json.dumps(chunk) + "\n\n").encode()) + self.wfile.write(b"data: [DONE]\n\n") + return + if self.path.endswith("/embeddings"): + response = { + "object": "list", + "model": CANARY, + "data": [{"object": "embedding", "index": 0, "embedding": [1.0, 2.0]}], + "usage": {"prompt_tokens": 5, "total_tokens": 5}, + } + else: + message = {"role": "assistant", "content": CANARY} + if model == "tools": + message = { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": CANARY + str(i), + "type": "function", + "function": { + "name": "probe.tool", + "arguments": json.dumps({"value": CANARY}), + }, + } + for i in range(2) + ], + } + if model == "journey": + completed = sum(item["role"] == "tool" for item in request["messages"]) + actions = [ + [("retrieve", {"query": CANARY, "mode": "semantic"})], + [ + ( + "get_entities", + { + "entities": [ + {"type": "block", "id": 11}, + {"type": "relation", "id": 21}, + {"type": "block", "id": 99}, + ] + }, + ) + ], + [ + ( + "resolver", + { + "action": "invoke", + "calls": [{"block_id": 11, "method": "get_text", "arguments": {}}], + }, + ) + ], + [ + ( + "submit_query_result", + {"answer": CANARY, "references": [{"type": "block", "id": 11}]}, + ), + ( + "submit_query_result", + {"answer": CANARY, "references": [{"type": "relation", "id": 21}]}, + ), + ], + ][completed] + message = { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": CANARY + str(index), + "type": "function", + "function": {"name": name, "arguments": json.dumps(arguments)}, + } + for index, (name, arguments) in enumerate(actions) + ], + } + response = { + "id": CANARY, + "object": "chat.completion", + "created": 1, + "model": CANARY, + "choices": [ + { + "index": 0, + "message": message, + "finish_reason": "tool_calls" if model == "tools" else CANARY, + } + ], + "usage": usage if model != "unknown" else None, + } + if model == "partial": + response["usage"] = {"prompt_tokens": -1, "completion_tokens": 2, "total_tokens": 2} + self.wfile.write(json.dumps(response).encode()) + + +class ToolInput(pydantic.BaseModel): + value: str + + +async def main(): + server = ThreadingHTTPServer(("127.0.0.1", 0), Server) + threading.Thread(target=server.serve_forever, daemon=True).start() + origin = f"http://127.0.0.1:{server.server_port}" + for signal in ("TRACES", "METRICS", "LOGS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = f"{origin}/otlp/{signal.lower()}" + start_telemetry(enabled=True, service_version="ai-probe") + names = { + 1: "known", + 2: "unknown", + 3: "partial", + 4: "stream", + 5: "tools", + 6: "error", + 7: "journey", + } + original_loader = AIManager.__dict__["_load_target_async"] + + async def load_target(cls, model_id): + dialect = ( + "core.alibaba-model-studio.v1" if model_id == 4 else "core.openai-compatible.v1" + ) + provider = AIProviderModel( + id=1, + name=CANARY, + dialect=dialect, + config={"api_key": CANARY, "base_url": origin + "/v1"}, + ) + model = AIModelModel( + id=model_id, + provider=1, + native_model_id=names[model_id], + capabilities=( + ChatCapability( + input_modalities=["text"], output_modalities=["text"], features=["tool_calling"] + ), + EmbeddingCapability(input_modalities=["text"], output_modalities=["vector"]), + ), + ) + return cls._execution_target(model_id, model, provider) + + AIManager._load_target_async = classmethod(load_target) + try: + user = UserMessage(content=(TextContentPart(text=CANARY),)) + for model_id in (1, 2, 3, 4): + result = await AIManager.chat(model_id, [user]) + assert result.content == CANARY + assert await AIManager.embed(1, [CANARY], 2) == ((1.0, 2.0),) + try: + await AIManager.chat(6, [user]) + except Exception as error: + assert CANARY in str(error) + else: + raise AssertionError("expected provider rejection") + + running = 0 + overlapped = asyncio.Event() + + async def handler(value): + nonlocal running + running += 1 + ordinal = running + if running == 2: + overlapped.set() + await asyncio.wait_for(overlapped.wait(), 2) + if ordinal == 2: + raise ToolExecutionError({"error": CANARY}) + return value.value + + backend = InMemoryThreadPersistenceBackend() + tool = BoundAgentTool( + FunctionTool( + id="probe.tool", description=CANARY, input_schema=ToolInput.model_json_schema() + ), + ToolInput, + handler, + ) + ident, state = await backend.create( + ThreadState( + model=5, + tools=(tool.definition,), + tool_choice="auto", + max_model_calls_per_turn=1, + messages=(), + ) + ) + thread = Thread(ident, state, backend, (tool,)) + assert (await thread.start_turn(user)).value == "max_model_calls" + assert overlapped.is_set() + assert len(thread.messages[-1].results) == 2 + assert [result.is_error for result in thread.messages[-1].results] == [False, True] + entered = asyncio.Event() + + async def blocking_handler(value): + entered.set() + await asyncio.Event().wait() + + blocking_tool = BoundAgentTool(tool.definition, ToolInput, blocking_handler) + ident, state = await backend.create(state) + cancelled_thread = Thread(ident, state, backend, (blocking_tool,)) + task = cancelled_thread.start_turn(user) + await asyncio.wait_for(entered.wait(), 2) + task.cancel() + try: + await task + except asyncio.CancelledError: + pass + else: + raise AssertionError("cancellation must propagate") + assert len(cancelled_thread.messages) == 1 + + # Synthetic domain data isolates the real Tool / Resolver / Thread / Sink journey. + block = BlockModel(id=11, resolver=TextResolver.__rsotype__, content=CANARY) + candidate_only = BlockModel(id=12, resolver=TextResolver.__rsotype__, content=CANARY) + relation = RelationModel(id=21, from_=11, to_=12, content=CANARY) + + async def retrieve_data(cls, query, mode, limit): + return { + "semantic": RetrievalBranch( + result=SemanticRetrievalResult( + profile=1, + matches=( + BlockSemanticRetrievalMatch(entity=block, score=0.9), + BlockSemanticRetrievalMatch(entity=candidate_only, score=0.8), + RelationSemanticRetrievalMatch(entity=relation, score=0.7), + ), + ) + ) + } + + async def read_data(cls, references, *, random_count=1): + assert references == (("block", 11), ("relation", 21), ("block", 99)) + return block, relation, None + + async def read_block(cls, identity): + assert identity == 11 + return block + + async def run_agent(cls, agent_id, initial_message): + bound = cls._bind_tools( + ( + info_tools.RETRIEVE_TOOL, + info_tools.GET_ENTITIES_TOOL, + resolver_tools.RESOLVER_TOOL, + "submit_query_result", + ) + ) + ident, state = await backend.create( + ThreadState( + model=7, + tools=tuple(tool.definition for tool in bound), + tool_choice="auto", + max_model_calls_per_turn=4, + messages=(), + ) + ) + thread = Thread(ident, state, backend, bound) + thread.start_turn(initial_message) + return thread + + saved = [ + (owner, name, owner.__dict__[name]) + for owner, name in [ + (InfoBaseManager, "retrieve"), + (InfoBaseManager, "get_entities"), + (BlockService, "get"), + (AgentManager, "run"), + ] + ] + InfoBaseManager.retrieve = classmethod(retrieve_data) + InfoBaseManager.get_entities = classmethod(read_data) + BlockService.get = classmethod(read_block) + AgentManager.run = classmethod(run_agent) + try: + job = JobModel( + id=31, type="core.sink.agent-query.v1", parameters={}, state={}, timeout_seconds=30 + ) + sink = AgentQuerySink( + SinkModel(id=41, type="core.agent-query.v1", config={"agent": 1}) + ) + with operation("job.execute", attributes={"inkcre.job.id": 31}): + await sink.execute(job, CANARY) + assert job.state["result"] == { + "answer": CANARY, + "references": [{"type": "relation", "id": 21}], + } + finally: + for owner, name, original in saved: + setattr(owner, name, original) + + finally: + await close_telemetry() + try: + outside_provider = TracerProvider() + with outside_provider.get_tracer("third-party").start_as_current_span( + "outside" + ) as outside: + assert (await AIManager.chat(4, [user])).content == CANARY + assert trace.get_current_span() is outside + assert not outside.attributes + outside_provider.shutdown() + finally: + AIManager._load_target_async = original_loader + server.shutdown() + + assert requests and exports + assert all(CANARY.encode() not in body for _, body in exports) + spans = [] + metrics = [] + logs = [] + for path, body in exports: + if path.endswith("traces"): + batch = ExportTraceServiceRequest.FromString(body) + spans.extend( + span + for resource in batch.resource_spans + for scope in resource.scope_spans + for span in scope.spans + ) + elif path.endswith("logs"): + batch = ExportLogsServiceRequest.FromString(body) + logs.extend( + record + for resource in batch.resource_logs + for scope in resource.scope_logs + for record in scope.log_records + ) + else: + batch = ExportMetricsServiceRequest.FromString(body) + metrics.extend( + metric + for resource in batch.resource_metrics + for scope in resource.scope_metrics + for metric in scope.metrics + ) + + def attrs(span): + return {item.key: item.value for item in span.attributes} + + chats = {} + for span in spans: + if span.name == "ai.chat": + chats.setdefault(attrs(span)["inkcre.ai.model.id"].int_value, span) + assert attrs(chats[1])["gen_ai.usage.output_tokens"].int_value == 0 + assert "gen_ai.usage.input_tokens" not in attrs(chats[2]) + assert "gen_ai.usage.input_tokens" not in attrs(chats[3]) + assert attrs(chats[3])["gen_ai.usage.output_tokens"].int_value == 2 + assert attrs(chats[4])["gen_ai.usage.input_tokens"].int_value == 7 + assert attrs(chats[4])["gen_ai.response.time_to_first_chunk"].double_value >= 0 + assert chats[6].status.code == 2 and not chats[6].status.message + assert all(not span.events for span in spans) + turn = next(span for span in spans if span.name == "agent.turn") + step = next(span for span in spans if span.name == "agent.step") + tools = [ + span for span in spans if span.name == "agent.tool" and span.trace_id == turn.trace_id + ] + assert step.parent_span_id == turn.span_id + assert chats[5].parent_span_id == step.span_id + assert len(tools) == 2 and all(span.parent_span_id == step.span_id for span in tools) + assert ( + tools[0].start_time_unix_nano < tools[1].end_time_unix_nano + and tools[1].start_time_unix_nano < tools[0].end_time_unix_nano + ) + assert sorted(span.status.code for span in tools) == [0, 2] + cancelled = [ + span + for span in spans + if attrs(span).get("inkcre.agent.outcome") + and attrs(span)["inkcre.agent.outcome"].string_value == "cancelled" + ] + assert {span.name for span in cancelled} == {"agent.turn", "agent.tool"} + assert all(span.status.code == 2 for span in cancelled) + tokens = next(metric for metric in metrics if metric.name == "inkcre.ai.token.usage") + totals = { + tuple((a.key, a.value.string_value) for a in point.attributes): point.as_int + for point in tokens.sum.data_points + } + chat_input = next( + value + for labels, value in totals.items() + if ("operation", "chat") in labels and ("token.type", "input") in labels + ) + assert chat_input == 56, ( + totals + ) # known + streaming + two Agent calls; repeated stream total is not accumulated. + assert all( + "model" not in attr.key and "tool" not in attr.key + for metric in metrics + for point in metric.sum.data_points + for attr in point.attributes + ) + assert [record.body.string_value for record in logs] == [ + "inkcre.retrieval.candidates", + "inkcre.entity.read", + "inkcre.entity.read", + "inkcre.agent.query.result", + ] + + def ids(record, field): + return tuple(item.int_value for item in attrs(record)[field].array_value.values) + + assert ids(logs[0], "inkcre.entity.block_ids") == (11, 12) + assert ids(logs[1], "inkcre.entity.block_ids") == (11,) + assert ids(logs[2], "inkcre.entity.block_ids") == (11,) + assert ids(logs[3], "inkcre.entity.block_ids") == () + assert ids(logs[3], "inkcre.entity.relation_ids") == (21,) + assert attrs(logs[3])["inkcre.agent.reference_count"].int_value == 1 + assert all(record.trace_id and record.span_id for record in logs) + assert len({record.trace_id for record in logs}) == 1 + print( + json.dumps( + { + "status": "passed", + "provider_http_requests": len(requests), + "otlp_batches": len(exports), + "spans": len(spans), + "source_events": len(logs), + "metric_names": [metric.name for metric in metrics], + "checks": [ + "unknown-zero-partial", + "usage-only-terminal-chunk", + "stream-total-once", + "embedding", + "concurrent-parentage", + "tool-error-result-preserved", + "cancellation-propagates", + "provider-error-content-absent", + "first-egress-canary-absent", + "disabled-preserves-third-party-context", + "candidate-read-final-reference-separation", + ], + }, + indent=2, + ) + ) + + +asyncio.run(main()) diff --git a/tasks/observability-foundation/experiments/application-probe.py b/tasks/observability-foundation/experiments/application-probe.py new file mode 100644 index 0000000..cdb77d3 --- /dev/null +++ b/tasks/observability-foundation/experiments/application-probe.py @@ -0,0 +1,197 @@ +"""Actual app lifespan and authenticated OTLP relay against the isolated task database.""" + +import importlib.util +import json +import os +from pathlib import Path +import signal +import subprocess +import sys +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +import httpx +import psycopg +from google.protobuf.json_format import MessageToDict +from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( + ExportTraceServiceRequest, +) + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +sys.path.insert(0, str(ROOT)) +spec = importlib.util.spec_from_file_location("db_lab", HERE / "implementation-db.py") +lab = importlib.util.module_from_spec(spec) +spec.loader.exec_module(lab) +os.environ.update(lab.environment()) +from app.middleware import create_peer_jwt + +records = [] +upstream_status = 200 +PRIVATE = "synthetic-relay-private-ingest" + + +class Receiver(BaseHTTPRequestHandler): + def do_POST(self): + data = self.rfile.read(int(self.headers.get("Content-Length", "0"))) + records.append((self.path, data, self.headers.get("Authorization"))) + self.send_response(upstream_status) + self.send_header("Content-Type", "application/x-protobuf") + self.send_header("Content-Length", "0") + self.end_headers() + + def log_message(self, *args): + pass + + +server = ThreadingHTTPServer(("127.0.0.1", 0), Receiver) +thread = threading.Thread(target=server.serve_forever, daemon=True) +thread.start() +base = "http://127.0.0.1:35501" +env = { + **lab.environment(), + "SKIP_EXTENSION_START": "1", + "OTEL_EXPORTER_OTLP_TIMEOUT": "1", + "PEER_ID": "00000000-0000-4000-8000-000000000002", + "OTEL_EXPORTER_OTLP_HEADERS": f"Authorization=Bearer%20{PRIVATE}", +} +for name in ("TRACES", "LOGS", "METRICS"): + env[f"OTEL_EXPORTER_OTLP_{name}_ENDPOINT"] = ( + f"http://127.0.0.1:{server.server_port}/v1/{name.lower()}" + ) +auth = { + "Authorization": "Bearer " + create_peer_jwt(lab.CREDENTIALS["jwt"]), + "Content-Type": "application/x-protobuf", +} +with psycopg.connect(lab.ADMIN) as connection: + deployment_id = connection.execute( + "SELECT value->>'deployment_id' FROM inkcre.configs WHERE key='inkcre.observability'" + ).fetchone()[0] +report = {} +try: + for enabled in (False, True): + before = len(records) + env["OBSRV__TELEMETRY_ENABLED"] = str(enabled).lower() + with (HERE / f"runtime/application-{enabled}.log").open("w") as log: + process = subprocess.Popen( + [ + sys.executable, + "-m", + "uvicorn", + "run:api_app", + "--host", + "127.0.0.1", + "--port", + "35501", + ], + cwd=ROOT, + env=env, + stdout=log, + stderr=subprocess.STDOUT, + ) + try: + with httpx.Client(base_url=base, timeout=4) as client: + deadline = time.monotonic() + 90 + while time.monotonic() < deadline: + if process.poll() is not None: + raise AssertionError("Application exited; inspect ignored runtime log") + try: + response = client.get("/readyz") + if response.status_code == 200: + break + except httpx.HTTPError: + pass + time.sleep(0.15) + else: + raise AssertionError("Application did not reach ready") + assert client.post("/telemetry/v1/traces", json={}).status_code == 401 + response = client.post("/telemetry/v1/traces", headers=auth, content=b"") + if not enabled: + assert response.status_code == 503 + else: + assert response.status_code == 200 and response.content == b"" + assert ( + client.post( + "/telemetry/v1/traces", + headers={**auth, "Content-Type": "application/json"}, + json={}, + ).status_code + == 415 + ) + assert ( + client.post( + "/telemetry/v1/traces", + headers=auth, + content=b"\xff", + ).status_code + == 400 + ) + assert ( + client.post( + "/telemetry/v1/traces", + headers=auth, + content=b"x" * (256 * 1024 + 1), + ).status_code + == 413 + ) + assert ( + client.post("/telemetry/v1/logs", headers=auth, content=b"").status_code + == 200 + ) + assert ( + client.post( + "/telemetry/v1/metrics", + headers={**auth, "Content-Type": "application/x-protobuf"}, + content=b"", + ).status_code + == 200 + ) + upstream_status = 204 + assert ( + client.post("/telemetry/v1/logs", headers=auth, content=b"").status_code + == 200 + ) + upstream_status = 503 + response = client.post("/telemetry/v1/traces", headers=auth, content=b"") + assert response.status_code == 502 and PRIVATE not in response.text + upstream_status = 200 + assert client.get("/readyz").status_code == 200 + report[str(enabled)] = { + "ready": True, + "unauthenticated_rejected": True, + "relay_status": 200 if enabled else 503, + } + finally: + started = time.monotonic() + process.send_signal(signal.SIGINT) + process.wait(timeout=40) + report[str(enabled)]["shutdown_seconds"] = round(time.monotonic() - started, 3) + assert process.returncode == 0 + if not enabled: + assert len(records) == before + assert records and all(header == f"Bearer {PRIVATE}" for _, _, header in records) + traces = [ + MessageToDict(ExportTraceServiceRequest.FromString(body)) + for path, body, _ in records + if path.endswith("/traces") + ] + rendered = json.dumps(traces) + assert PRIVATE not in rendered and lab.CREDENTIALS["jwt"] not in rendered + assert deployment_id in rendered + report["checks"] = [ + "default-off makes zero OTLP requests", + "real run.py readyz/lifespan both modes", + "existing Peer JWT required", + "protobuf preserved; authenticated JSON rejected with 415", + "private server authorization only upstream", + "400 malformed / 413 capacity / 502 upstream rejection", + "shared deployment identity present", + "no credentials in trace payload", + ] + (HERE / "evidence/application-probe.json").write_text(json.dumps(report, indent=2) + "\n") + print(json.dumps(report, indent=2)) +finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) diff --git a/tasks/observability-foundation/experiments/candidate-types.py b/tasks/observability-foundation/experiments/candidate-types.py new file mode 100644 index 0000000..12ae4e8 --- /dev/null +++ b/tasks/observability-foundation/experiments/candidate-types.py @@ -0,0 +1,73 @@ +"""Generate candidate client types from the task's migrated database using Supabase.""" + +import hashlib +import json +from pathlib import Path +import shlex +import subprocess + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +CLIENT = ROOT.parent / "client-web" +credentials = json.loads((HERE / "runtime/database-credential.json").read_text()) +url = ( + f"postgresql://postgres:{credentials['admin']}@127.0.0.1:35432/o11y_impl?sslmode=disable" +) +command = shlex.join( + [ + "npx", + "--yes", + "supabase@2.112.0", + "gen", + "types", + "typescript", + "--db-url", + url, + "--schema", + "inkcre", + ] +) +script = ( + "set -eu\nworkbin=$(mktemp -d)\n" + "trap 'rm -rf \"$workbin\"' EXIT\n" + "ln -s '/mnt/c/Program Files/Docker/Docker/resources/bin/docker.exe' " + '"$workbin/docker"\n' + 'PATH="$workbin:$PATH" ' + command + "\n" +) +result = subprocess.run( + ["ssh", "-T", "-o", "BatchMode=yes", "wsl.win-ws.localhost", "bash", "-s"], + input=script, + capture_output=True, + text=True, + check=False, + timeout=300, +) +if result.returncode: + error = result.stderr + for secret in credentials.values(): + error = error.replace(secret, "") + raise RuntimeError(error) +assert ( + "submission_traceparent" in result.stdout and "submission_tracestate" in result.stdout +) +formatted = subprocess.run( # noqa: S603 + [str(CLIENT / "node_modules/.bin/oxfmt"), "--stdin-filepath", "database.generated.ts"], + input=result.stdout, + cwd=CLIENT, + capture_output=True, + text=True, + check=True, +) +output = CLIENT / "packages/core/src/database/database.generated.ts" +output.write_text(formatted.stdout) +report = { + "source": "candidate migrated disposable o11y_impl, not production-admitted artifact", + "migration_head": "3d9593b0c855", + "generator": "supabase@2.112.0", + "sha256": hashlib.sha256(output.read_bytes()).hexdigest(), + "required_before_merge": ( + "regenerate/compare via sync-database-contract.mjs from admitted core artifact" + ), +} +(HERE / "runtime/candidate-types.json").write_text(json.dumps(report, indent=2) + "\n") +print(json.dumps(report, indent=2)) diff --git a/tasks/observability-foundation/experiments/carrier-capacity.py b/tasks/observability-foundation/experiments/carrier-capacity.py new file mode 100644 index 0000000..7f49bfc --- /dev/null +++ b/tasks/observability-foundation/experiments/carrier-capacity.py @@ -0,0 +1,48 @@ +# /// script +# requires-python = ">=3.12" +# dependencies = ["opentelemetry-sdk==1.45.0"] +# /// +"""Prove SDK normalization and the need for an application-owned persistence size +boundary.""" + +import json +from pathlib import Path + +from opentelemetry import trace +from opentelemetry.context import Context +from opentelemetry.trace import NonRecordingSpan, SpanContext, TraceFlags, TraceState +from opentelemetry.trace.propagation.tracecontext import TraceContextTextMapPropagator + +propagator = TraceContextTextMapPropagator() +state = TraceState([(f"v{i}", "s" * 30) for i in range(32)]) +source = SpanContext(1, 2, False, TraceFlags(1), state) +carrier = {} +propagator.inject(carrier, trace.set_span_in_context(NonRecordingSpan(source), Context())) +assert len(carrier["traceparent"]) == 55 +assert len(carrier["tracestate"].encode()) > 512 +# Candidate boundary policy: keep traceparent, omit oversized optional state as a whole. +# It does not parse W3C, truncate inside an entry, or fail the business create transaction. +bounded = {key: value for key, value in carrier.items() if len(value.encode()) <= 512} +restored = trace.get_current_span(propagator.extract(bounded, Context())).get_span_context() +assert (restored.trace_id, restored.span_id, restored.trace_flags) == ( + source.trace_id, + source.span_id, + source.trace_flags, +) +assert not restored.trace_state +future = {"traceparent": "01" + carrier["traceparent"][2:] + "-extension"} +normalized = {} +propagator.inject(normalized, propagator.extract(future, Context())) +assert normalized["traceparent"] == carrier["traceparent"] +report = { + "sdk_injected_traceparent_bytes": 55, + "sdk_injected_tracestate_bytes": len(carrier["tracestate"].encode()), + "candidate_max_bytes_each": 512, + "oversize_state_policy": "omit optional state as a whole; preserve parent IDs/flags", + "future_context_reinjected_as_version_00": True, + "persistence_limit_is_application_policy_not_W3C_maximum": True, +} +(Path(__file__).parent / "runtime/carrier-capacity.json").write_text( + json.dumps(report, indent=2) + "\n" +) +print(json.dumps(report, indent=2)) diff --git a/tasks/observability-foundation/experiments/client-browser-result.json b/tasks/observability-foundation/experiments/client-browser-result.json new file mode 100644 index 0000000..629988c --- /dev/null +++ b/tasks/observability-foundation/experiments/client-browser-result.json @@ -0,0 +1,32 @@ +{ + "passed": true, + "encoding": "application/x-protobuf", + "browser": "149.0.7827.55", + "jobIds": [ + 67, + 68, + 72 + ], + "spans": 74, + "signals": [ + "/v1/logs", + "/v1/traces", + "/v1/metrics" + ], + "independentJobLink": true, + "concurrentNativeAwaitIsolation": true, + "mixedPeers": [ + "off→on", + "on→off" + ], + "metadataSentinelsAbsent": true, + "crossOriginRedirectBlocked": true, + "unsampledCarrier": true, + "oversizedTracestateDropped": true, + "invalidCarrierIgnored": true, + "metricsOnlyTerminalOutcomes": true, + "lifecycleInitialization": true, + "relayJwtCapturedConnection": true, + "publicOtlpWithoutJwt": true, + "failedExporterElapsedMs": 3 +} diff --git a/tasks/observability-foundation/experiments/client-browser.mjs b/tasks/observability-foundation/experiments/client-browser.mjs new file mode 100644 index 0000000..288779a --- /dev/null +++ b/tasks/observability-foundation/experiments/client-browser.mjs @@ -0,0 +1,191 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { createServer as httpServer } from 'node:http' +import { readFile, writeFile } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { fileURLToPath } from 'node:url' +const client = fileURLToPath(new URL('../../../../client-web', import.meta.url)) +const require = createRequire(`${client}/package.json`) +const { chromium } = require('@playwright/test') +const { createServer } = await import(`${client}/apps/client-web/node_modules/vite/dist/node/index.js`) +const { jwtVerify } = await import(`${client}/apps/client-web/node_modules/jose/dist/webapi/index.js`) +const credential = JSON.parse(await readFile(new URL('./runtime/database-credential.json', import.meta.url))) +let captured = [] +const peerRequests = [] +let redirectTargetRequests = 0 +const relayAuthentication = [], publicAuthentication = [] +const redirectTarget = httpServer((_request, response) => { redirectTargetRequests += 1; response.setHeader('Access-Control-Allow-Origin', '*'); response.end('{}') }) +await new Promise(resolve => redirectTarget.listen(0, '127.0.0.1', resolve)) +const redirectUrl = `http://127.0.0.1:${redirectTarget.address().port}/external` +const sink = httpServer(async (request, response) => { + response.setHeader('Access-Control-Allow-Origin', '*') + response.setHeader('Access-Control-Allow-Headers', '*') + if (request.method === 'OPTIONS') { response.end(); return } + if (request.url === '/redirect') { response.writeHead(302, { Location: redirectUrl }); response.end(); return } + if (request.url === '/server-error') { response.writeHead(500); response.end('{}'); return } + if (request.url === '/not-executed') { response.writeHead(503, { 'InkCre-Peer-Execution': 'not-executed', 'Access-Control-Expose-Headers': 'InkCre-Peer-Execution' }); response.end('{}'); return } + if (request.url.startsWith('/peer')) { + peerRequests.push({ traceparent: request.headers.traceparent, tracestate: request.headers.tracestate }) + response.setHeader('Content-Type', 'application/json') + response.end('{}'); return + } + const chunks = [] + for await (const chunk of request) chunks.push(chunk) + const body = Buffer.concat(chunks) + if (request.url.startsWith('/relay/')) { + try { await jwtVerify(request.headers.authorization?.replace(/^Bearer /, ''), new TextEncoder().encode(credential.jwt)); relayAuthentication.push(true) } catch { relayAuthentication.push(false) } + } else publicAuthentication.push(Boolean(request.headers.authorization)) + captured.push({ signal: request.url.replace('/relay', ''), data: body.toString('base64') }) + response.setHeader('Content-Type', 'application/x-protobuf') + response.end() +}) +await new Promise(resolve => sink.listen(0, '127.0.0.1', resolve)) +const endpoint = `http://127.0.0.1:${sink.address().port}` +const vite = await createServer({ configFile: false, root: `${client}/packages/core`, server: { host: '127.0.0.1', port: 0, fs: { allow: [client] } }, optimizeDeps: { entries: [], include: ['vue', 'pinia', 'zod', 'zod-class', 'zod-config', 'jose', '@supabase/postgrest-js', '@opentelemetry/api', '@opentelemetry/core', '@opentelemetry/resources', '@opentelemetry/sdk-trace-web', '@opentelemetry/exporter-trace-otlp-proto', '@opentelemetry/sdk-logs', '@opentelemetry/exporter-logs-otlp-proto', '@opentelemetry/sdk-metrics', '@opentelemetry/exporter-metrics-otlp-proto'] } }) +vite.middlewares.use('/probe', (_req, res) => { res.setHeader('Content-Type', 'text/html'); res.end('InKCre synthetic browser observability acceptance') }) +await vite.listen() +const browser = await chromium.launch({ headless: true }) +try { + const page = await browser.newPage() + const errors = [] + page.on('pageerror', error => errors.push(error.message)) + await page.goto(`http://127.0.0.1:${vite.httpServer.address().port}/probe`) + const result = await page.evaluate(async ({ endpoint, client, jwt }) => { + const t = await import(`/src/obsrv/telemetry.ts`) + const { configStore } = await import(`/src/config/store.ts`) + const { DBAPIClient } = await import('/src/base/db-api.ts') + const { MetaConfigSchema } = await import(`/src/config/schema.ts`) + const { PeerHTTPOutbound } = await import(`/src/peer/http.ts`) + const { Job, JobType } = await import(`/src/job/job.ts`) + const { JobManager } = await import(`/src/job/manager.ts`) + const { z } = await import('/node_modules/.vite/deps/zod.js') + const { trace, createTraceState } = await import('/node_modules/.vite/deps/@opentelemetry_api.js') + const { PeerOutcomeUnknown, PeerRequestNotExecuted } = await import('/src/peer/contracts.ts') + const ensure = (condition, label) => { if (!condition) throw new Error(label) } + const peerId = '00000000-0000-4000-8000-000000000001' + configStore.metaConfig = MetaConfigSchema.parse({ INKCRE_PGREST_URL: 'http://127.0.0.1:33000', INKCRE_JWT_SECRET: jwt, INKCRE_PEER_ID: peerId }) + ensure(configStore.metaConfig.telemetry_enabled === false, 'default off') + await t.initializeTelemetry(configStore.metaConfig, 'synthetic') + ensure(!performance.getEntriesByType('resource').some(entry => new URL(entry.name).pathname.endsWith('/configs')), 'disabled does not read shared config') + const offCarrier = t.captureSubmission(t.ROOT_CONTEXT) + ensure(offCarrier.submission_traceparent === null, 'disabled carrier') + const types = await JobType.getAll() + const type = types[0].id + let executions = 0 + JobManager.registerHandler(type, { parameters: z.record(z.string(), z.unknown()), canHandle: () => true, async handle(job, parameters, signal) { await new Promise(resolve => setTimeout(resolve, 8)); executions += 1; if (parameters.mode === 'error') throw new Error('PRIVATE_EXCEPTION_SENTINEL'); if (parameters.mode === 'timeout' || parameters.mode === 'abort') await new Promise(resolve => { if (signal.aborted) resolve(); else signal.addEventListener('abort', resolve, { once: true }) }); job.state = { synthetic: true } } }) + const offJob = await JobManager.create(type, { sentinel: 'PRIVATE_BODY_SENTINEL' }) + ensure(offJob.submission_traceparent === null, 'off database defaults') + const configDb = new DBAPIClient('configs') + const { data: previousConfig } = await configDb.from().select('schema,value').eq('key', 'inkcre.observability').single().throwOnError() + const cfg = { deployment_id: previousConfig.value.deployment_id, otlp_http_endpoints: { traces: `${endpoint}/v1/traces`, logs: `${endpoint}/v1/logs`, metrics: `${endpoint}/v1/metrics` }, diagnostics_url: `${endpoint}/diagnostics` } + await configDb.update({ value: cfg }).eq('key', 'inkcre.observability').throwOnError() + try { + configStore.metaConfig.telemetry_enabled = true + await t.initializeTelemetry(configStore.metaConfig, 'synthetic') + ensure(t.telemetryDiagnosticsUrl.value === cfg.diagnostics_url, 'real lifecycle reads config') + const parents = await Promise.all([25, 4].map(delay => t.observeOperation('job.submit', async active => { + await new Promise(resolve => setTimeout(resolve, delay)) + const parent = t.captureSubmission(active).submission_traceparent + const outbound = new PeerHTTPOutbound({ id: peerId }, { method: 'POST', url: `${endpoint}/peer` }) + await outbound.execute({ query: { private: ['PRIVATE_QUERY_SENTINEL'] }, body: { secret: 'PRIVATE_BODY_SENTINEL' } }, active) + return parent + }))) + ensure(parents[0].split('-')[1] !== parents[1].split('-')[1], 'concurrent context isolated') + const redirect = new PeerHTTPOutbound({ id: peerId }, { method: 'POST', url: `${endpoint}/redirect` }) + try { await redirect.execute({}); throw new Error('redirect unexpectedly followed') } catch (error) { ensure(error instanceof PeerOutcomeUnknown, 'redirect retains unknown outcome') } + const rejected = new PeerHTTPOutbound({ id: peerId }, { method: 'POST', url: `${endpoint}/not-executed` }) + try { await rejected.execute({}); throw new Error('non-execution unexpectedly accepted') } catch (error) { ensure(error instanceof PeerRequestNotExecuted, 'not-executed retained') } + const largeState = Array.from({ length: 8 }).reduce((state, _, i) => state.set(`vendor${i}`, 'a'.repeat(140)), createTraceState()) + const carrierContext = trace.setSpanContext(t.ROOT_CONTEXT, { traceId: 'a'.repeat(32), spanId: 'b'.repeat(16), traceFlags: 0, traceState: largeState }) + const bounded = t.captureSubmission(carrierContext) + ensure(bounded.submission_traceparent?.endsWith('-00') && bounded.submission_tracestate === null, 'unsampled accepted and oversized optional state omitted') + ensure(t.submissionLinks(bounded).links[0].context.traceFlags === 0, 'unsampled link flags retained') + ensure(!t.submissionLinks({ submission_traceparent: 'PRIVATE_INVALID_CARRIER' }).links, 'invalid carrier ignored by SDK') + const onJob = await JobManager.create(type, { sentinel: 'PRIVATE_BODY_SENTINEL' }) + ensure(onJob.submission_traceparent?.length === 55, 'real job carrier persisted') + const submittedCarrier = onJob.submission_traceparent + await JobManager.run(onJob.id) + const finished = await Job.get(onJob.id) + ensure(finished.status === 'finished' && finished.submission_traceparent === submittedCarrier, 'claim close preserve carrier') + await JobManager.run(offJob.id) + const oldFinished = await Job.get(offJob.id) + ensure(oldFinished.status === 'finished' && oldFinished.submission_traceparent === null, 'off producer on consumer') + async function runFailures() { + const serverError = new PeerHTTPOutbound({ id: peerId }, { method: 'POST', url: `${endpoint}/server-error` }) + const response = await serverError.execute({}) + ensure(response.status === 500, 'readable 500 business envelope retained') + const failed = await JobManager.create(type, { mode: 'error' }) + await JobManager.run(failed.id) + ensure((await Job.get(failed.id)).status === 'failed', 'failure result retained') + const timeout = await JobManager.create(type, { mode: 'timeout' }, 1) + await JobManager.run(timeout.id) + ensure((await Job.get(timeout.id)).status === 'timed_out', 'timeout result retained') + const aborted = await JobManager.create(type, { mode: 'abort' }) + const running = JobManager.run(aborted.id) + while ((await Job.get(aborted.id)).status === 'pending') await new Promise(resolve => setTimeout(resolve, 10)) + await new Promise(resolve => setTimeout(resolve, 20)) + await Job.dbApi.update({ abort_requested: true }).eq('id', aborted.id).throwOnError() + await JobManager.checkAbortRequests() + await running + ensure((await Job.get(aborted.id)).status === 'aborted', 'abort result retained') + } + await runFailures() + const onOff = await JobManager.create(type, {}) + await t.flushTelemetry() + await t.shutdownTelemetry() + await JobManager.run(onOff.id) + const mixed = await Job.get(onOff.id) + ensure(mixed.status === 'finished' && mixed.submission_traceparent === onOff.submission_traceparent, 'on producer off consumer preserves carrier') + const offOutbound = new PeerHTTPOutbound({ id: peerId }, { method: 'POST', url: `${endpoint}/peer` }) + await offOutbound.execute({}) + await t.configureTelemetry({ ...cfg, otlp_http_endpoints: { metrics: `${endpoint}/v1/metrics` } }, peerId, 'metrics-only') + await runFailures() + await t.flushTelemetry() + await t.shutdownTelemetry() + configStore.metaConfig.telemetry_peer_relay_url = `${endpoint}/relay` + await t.initializeTelemetry(configStore.metaConfig, 'relay') + configStore.metaConfig.INKCRE_JWT_SECRET = 'different-connection-synthetic-secret-for-snapshot-probe' + await t.observeOperation('job.submit', async () => true) + await t.flushTelemetry() + await t.shutdownTelemetry() + configStore.metaConfig.INKCRE_JWT_SECRET = jwt + const started = performance.now() + await t.configureTelemetry({ ...cfg, otlp_http_endpoints: { traces: 'http://127.0.0.1:1/v1/traces' } }, peerId, 'synthetic') + await t.observeOperation('job.submit', async () => true) + await t.flushTelemetry() + await t.shutdownTelemetry() + ensure(performance.now() - started < 3000, 'failed exporter bounded') + return { parents, jobs: [offJob.id, onJob.id, onOff.id], submittedCarrier, executions, failedExporterElapsedMs: Math.round(performance.now() - started) } + } finally { configStore.metaConfig.INKCRE_JWT_SECRET = jwt; await t.shutdownTelemetry(); await configDb.update(previousConfig).eq('key', 'inkcre.observability').throwOnError() } + }, { endpoint, client, jwt: credential.jwt }) + captured = JSON.parse(execFileSync('pdm', ['run', 'python', 'tasks/observability-foundation/experiments/client-decode-protobuf.py'], { cwd: `${client}/../core-py`, input: JSON.stringify(captured), encoding: 'utf8' })) + assert.equal(errors.length, 0, errors.join('\n')) + assert.equal(result.executions, 9) + const traceBatches = captured.filter(item => item.signal === '/v1/traces') + const spans = traceBatches.flatMap(batch => batch.data.resourceSpans.flatMap(resource => resource.scopeSpans.flatMap(scope => scope.spans))) + const attr = (span, key) => span.attributes?.find(a => a.key === key)?.value + const execution = spans.find(span => span.name === 'job.execute' && String(attr(span, 'inkcre.job.id')?.intValue) === String(result.jobs[1])) + assert(execution, 'job execution span') + assert.equal(execution.links[0].traceId, result.submittedCarrier.split('-')[1]) + assert.notEqual(execution.traceId, execution.links[0].traceId) + assert(!execution.parentSpanId, 'independent execution root') + assert(relayAuthentication.length >= 3 && relayAuthentication.every(Boolean), 'relay dynamic JWT uses captured connection') + assert(publicAuthentication.every(value => value === false), 'public OTLP receives no Peer JWT') + assert.equal(redirectTargetRequests, 0, 'cross-origin redirect receives no request or trace context') + assert.equal(peerRequests.length, 3) + for (const parent of result.parents) assert(peerRequests.some(request => request.traceparent?.split('-')[1] === parent.split('-')[1])) + assert.equal(peerRequests[2].traceparent, undefined) + for (const signal of ['traces', 'logs', 'metrics']) assert(captured.some(item => item.signal === `/v1/${signal}`), `${signal} exported`) + const serialized = JSON.stringify(captured) + assert(!serialized.includes('PRIVATE_BODY_SENTINEL') && !serialized.includes('PRIVATE_QUERY_SENTINEL') && !serialized.includes('PRIVATE_EXCEPTION_SENTINEL')) + const metricResources = captured.filter(item => item.signal === '/v1/metrics').flatMap(batch => batch.data.resourceMetrics) + for (const version of ['synthetic', 'metrics-only']) { + const outcomes = metricResources.filter(resource => resource.resource.attributes.some(a => a.key === 'service.version' && a.value.stringValue === version)).flatMap(resource => resource.scopeMetrics.flatMap(scope => scope.metrics.filter(metric => metric.name === 'inkcre.operation.duration').flatMap(metric => metric.histogram.dataPoints))).filter(point => point.attributes.some(a => a.key === 'inkcre.operation' && a.value.stringValue === 'job.execute')).map(point => point.attributes.find(a => a.key === 'inkcre.outcome').value.stringValue) + for (const outcome of ['error', 'timeout', 'cancelled']) assert(outcomes.includes(outcome), `${version} ${outcome} measured independently of trace export`) + const httpPoints = metricResources.filter(resource => resource.resource.attributes.some(a => a.key === 'service.version' && a.value.stringValue === version)).flatMap(resource => resource.scopeMetrics.flatMap(scope => scope.metrics.filter(metric => metric.name === 'inkcre.operation.duration').flatMap(metric => metric.histogram.dataPoints))) + assert(httpPoints.some(point => point.attributes.some(a => a.key === 'inkcre.operation' && a.value.stringValue === 'peer.http') && point.attributes.some(a => a.key === 'inkcre.outcome' && a.value.stringValue === 'error')), `${version} readable HTTP 500 counted as error`) + } + const report = { passed: true, encoding: 'application/x-protobuf', browser: await browser.version(), jobIds: result.jobs, spans: spans.length, signals: [...new Set(captured.map(item => item.signal))], independentJobLink: true, concurrentNativeAwaitIsolation: true, mixedPeers: ['off→on', 'on→off'], metadataSentinelsAbsent: true, crossOriginRedirectBlocked: true, unsampledCarrier: true, oversizedTracestateDropped: true, invalidCarrierIgnored: true, metricsOnlyTerminalOutcomes: true, lifecycleInitialization: true, relayJwtCapturedConnection: true, publicOtlpWithoutJwt: true, failedExporterElapsedMs: result.failedExporterElapsedMs } + await writeFile(new URL('./client-browser-result.json', import.meta.url), JSON.stringify(report, null, 2) + '\n') + console.log(JSON.stringify(report)) +} finally { await browser.close(); await vite.close(); await new Promise(resolve => sink.close(resolve)); await new Promise(resolve => redirectTarget.close(resolve)) } diff --git a/tasks/observability-foundation/experiments/client-decode-protobuf.py b/tasks/observability-foundation/experiments/client-decode-protobuf.py new file mode 100644 index 0000000..540053d --- /dev/null +++ b/tasks/observability-foundation/experiments/client-decode-protobuf.py @@ -0,0 +1,39 @@ +"""Inspect local OTLP protobuf capture using the official generated message classes.""" + +import base64 +import json +import sys +from google.protobuf.json_format import MessageToDict +from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ExportLogsServiceRequest +from opentelemetry.proto.collector.metrics.v1.metrics_service_pb2 import ( + ExportMetricsServiceRequest, +) +from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( + ExportTraceServiceRequest, +) + +models = { + "/v1/traces": ExportTraceServiceRequest, + "/v1/logs": ExportLogsServiceRequest, + "/v1/metrics": ExportMetricsServiceRequest, +} + + +def inspect(value): + if isinstance(value, list): + return [inspect(item) for item in value] + if isinstance(value, dict): + return { + key: base64.b64decode(item).hex() + if key in {"traceId", "spanId", "parentSpanId"} + else inspect(item) + for key, item in value.items() + } + return value + + +records = json.load(sys.stdin) +for record in records: + model = models[record["signal"]].FromString(base64.b64decode(record["data"])) + record["data"] = inspect(MessageToDict(model)) +json.dump(records, sys.stdout) diff --git a/tasks/observability-foundation/experiments/client-implementation.md b/tasks/observability-foundation/experiments/client-implementation.md new file mode 100644 index 0000000..237567f --- /dev/null +++ b/tasks/observability-foundation/experiments/client-implementation.md @@ -0,0 +1,85 @@ +# client-web 第一批观测实现与验收 + +2026-10-03,分支 `codex/observability-foundation`。此报告属于父任务,不建立第二控制权。 +子工作未 commit/push,未改 shared 挂载或生成数据库协议文件;候选生成类型由 primary 写入。 + +## 交付面 + +`packages/core/src/obsrv/telemetry.ts` 集中拥有标准 OTel SDK 的三信号初始化、公开配置校验、 +有界排空、W3C SDK carrier 与 Span Link、固定元数据事件和低基数操作耗时。 +连接本地 `telemetry_enabled=false` 是唯一启用开关。关闭不读共享观测配置、不启 provider, +默认 Job INSERT carrier 为 null。没有出口时保留固定原因本地诊断而不启 SDK。 + +`telemetry_peer_relay_url` 为空时使用共享完整 per-signal public URL 且不发送 Peer JWT; +显式设置时追加 `/v1/traces`、`/v1/logs`、`/v1/metrics`。SDK 的标准 async headers factory +通过现有签名 authority 生成短期 Peer JWT,捕获初始化时的连接,不能把旧批次绑定到后来 +切换的 secret。没有 SaaS 私密 token 字段,没有隐式 relay 推导、失败回退或新增常驻服务。 + +PeerManager 的异步选择 context 显式传给 PeerHTTPOutbound,后者负责一次 W3C 注入。 +开启态拒绝所有 Peer 重定向以守住配置 endpoint 边界,错误仍是原来的 +`PeerOutcomeUnknown` 且不重试;关闭态保留原 fetch 跟随行为。PostgREST 仅作本地 client +span,不注入 W3C,也不声称有 server/SQL span。无全局 fetch/Promise patch。 + +Job 创建同 INSERT 写 carrier;执行成功 claim 后独立 trace Link 提交,所有 claim 分支 +共用执行观测。claim/close 不覆盖 submission 列。`job.submitted`、`job.started`、 +`job.closed` 事件来自持久化边界。失败、超时、取消的 metric outcome 独立于 trace 导出状态, +不会因捕获业务错误后返回而记成功。原 PG 日志、任意业务错误 message 和结果 state 保留原 +路径,不桥接到新增 OTLP。Job 页仅增加可选诊断基础链接,以 Job ID 查询。 + +SPA 生命周期、设置保存/导入、重置和关闭已接入。设置控件使用安装的 UI 2.2.1 规范。 +默认关闭和空 relay 已在生产构建浏览器页面验证,截图见 `client-settings.png`。 + +## 实际验收 + +执行:在 client-web 下运行 `pnpm exec node ../core-py/tasks/observability-foundation/experiments/client-browser.mjs`。 +脚本依赖 parent 创建的 disposable `o11y_impl` PostgreSQL/PostgREST(loopback 33000), +只从忽略的 0600 `runtime/database-credential.json` 读取合成 JWT secret,不输出它。 +它临时改写公共 `inkcre.observability` 出口,保留 deployment ID,并在 finally 恢复原 +schema/value;运行前须与 parent 协调该 key 的独占使用。不能对共享 dev/preview 运行。 + +最新 `client-browser-result.json` 记录 Chromium 149.0.7827.55、74 spans 和三个 OTLP/HTTP protobuf +信号。合成 Job 主样本 67、68、72,另有错误/取消/超时样本。实际断言覆盖: + +- 关闭态不读取共享配置,Job carrier 为 null;正常 lifecycle 开启读取公共配置。 +- 两条并发原生 await 链显式 context 隔离,Peer HTTP 实际收到正确 traceparent。 +- 真实 Job INSERT、claim、close 和 off→on / on→off 组合保留列与终态,执行独立 root Link。 +- 无效 carrier 被 SDK 忽略,未采样标志保留;SDK API set 构造超过 512 bytes 的 tracestate 后,应用省略可选字段。 +- 双 origin 真 HTTP 重定向第二端收到零请求,not-executed 与可读 HTTP 500 的业务语义不变。 +- 完整三信号和 metrics-only 的 Job error/timeout/cancelled 与 HTTP 500 error 标签一致。 +- 新增 OTLP 不含请求 body/query/异常 message 的合成 sentinel,公开出口不带 Peer JWT。 +- 中转三信号的动态 JWT 通过真实密码学验证;修改全局连接 secret 后旧 provider 仍用捕获的原 secret。 +- 不可达出口不影响业务返回,flush/shutdown 有界;页面退出仍只 best effort。 + +这使用真实浏览器、SDK、PostgREST 和本地 OTLP HTTP 接收端,不冒充 Grafana Cloud +账号、远端保留、查询或限额已通过。真实 core-py relay 联调见下节。 + +## 真实 core relay 联调 + +`client-real-relay.mjs` 在真实 Chromium 中读取未改写的共享配置,使用官方 protobuf +exporter 发送合成元数据到 `http://127.0.0.1:35501/telemetry`。relay 来自实际 core-py +应用进程,只有其上游是本地合成 OTLP 接收端。`client-real-relay-verify.py` 通过标准 +生成的 protobuf 消息解析上游落盘数据,结果在 `client-real-relay-result.json`。 + +三信号的 SDK protobuf 请求和响应均为 200;缺失/无效 Peer JWT 均 401,带合法 JWT 的 +JSON 请求为 415。接收端证实服务器私密合成 ingest 认证正确,浏览器没有持有它。 +提交与执行两个 Trace ID、两个 Span ID、独立执行 trace 的提交 Link 原值逐项一致; +Link sampled 标志保留,两个 Job 事件的 Trace/Span ID 与 Job ID 正确,metric label +仅操作名与结果,资源部署/Peer/实例身份齐全。该脚本不改共享配置、不使用 Cloud。 + +联调实际捕获并修复了两个缺陷:Python 的可选配置字段会序列化为 null,TS 现已同时 +接受缺省与 null;泛用 ProtoJSON 解析会把 OTLP JSON 的 hex ID 当作 Base64,曾生成 +错误长度的 ID。最终选择官方 OTLP/HTTP protobuf exporter,core relay 仅接受 protobuf, +删除该 JSON 转换责任,并用精确 ID 与 Link 比较替代仅检查 HTTP 200。 + +## 检查与剩余边界 + +`pnpm lint`、全 workspace `pnpm type-check`、`pnpm build`、portable package check 均通过。 +所有 tracked 文件及新增 telemetry.ts 的 oxfmt 检查和 `git diff --check` 通过。 +完整 `pnpm check` 在格式阶段被用户原有未跟踪 +`.agents/skills/vue-ts-code/{SKILL.md,scripts/audit.mjs}` 阻塞;按委派要求保留原状,未改或删除。 + +浏览器扩展、第三方 native extension handler 内部和任意 provider 原生 await 不自动继承 +Job 执行 context;首批不改全仓 target、不引 Zone、不声称完整执行调用树。 +Webext 仅获得共用配置的默认关闭字段,尚未接入其独立宿主遥测生命周期。 +发布前需从正式获准 core artifact 重跑既有 database contract 同步;本轮生成列是 primary +从 candidate migration `3d9593b0c855` 的实际隔离 schema 生成,不能冒充 stable release 已准入。 diff --git a/tasks/observability-foundation/experiments/client-real-relay-result.json b/tasks/observability-foundation/experiments/client-real-relay-result.json new file mode 100644 index 0000000..88d52d9 --- /dev/null +++ b/tasks/observability-foundation/experiments/client-real-relay-result.json @@ -0,0 +1,55 @@ +{ + "passed": true, + "browser": "149.0.7827.55", + "relay": "http://127.0.0.1:35501/telemetry", + "version": "browser-real-relay-protobuf-v1", + "jwtMissingStatus": 401, + "jwtInvalidStatus": 401, + "accepted": [ + { + "signal": "traces", + "status": 200, + "responseType": "application/x-protobuf", + "inputType": "application/x-protobuf", + "authenticated": true + }, + { + "signal": "logs", + "status": 200, + "responseType": "application/x-protobuf", + "inputType": "application/x-protobuf", + "authenticated": true + }, + { + "signal": "metrics", + "status": 200, + "responseType": "application/x-protobuf", + "inputType": "application/x-protobuf", + "authenticated": true + }, + { + "signal": "metrics", + "status": 200, + "responseType": "application/x-protobuf", + "inputType": "application/x-protobuf", + "authenticated": true + } + ], + "traceId": "12b067ca55a4090914f0cbbe1d9fac8b", + "spanId": "56d6dc557b8fc993", + "executionTraceId": "77ee1ff13e144b1278e77b046bac3b2d", + "executionSpanId": "1572af5b7c20faee", + "unsupportedJsonStatus": 415, + "sharedConfigMutated": false, + "cloudExportUsed": false, + "upstreamVerified": true, + "upstreamBrowserResources": { + "/v1/traces": 1, + "/v1/logs": 1, + "/v1/metrics": 2 + }, + "upstreamServerAuthentication": true, + "relayEncoding": "OTLP/protobuf → OTLP/protobuf", + "exactTraceSpanIdsAndLinkPreserved": true, + "correlatedJobEventsAndBoundedMetricLabels": true +} diff --git a/tasks/observability-foundation/experiments/client-real-relay-verify.py b/tasks/observability-foundation/experiments/client-real-relay-verify.py new file mode 100644 index 0000000..7e2a52b --- /dev/null +++ b/tasks/observability-foundation/experiments/client-real-relay-verify.py @@ -0,0 +1,115 @@ +"""Verify synthetic browser batches captured beyond the real core OTLP relay.""" + +import json +from pathlib import Path +from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ExportLogsServiceRequest +from opentelemetry.proto.collector.metrics.v1.metrics_service_pb2 import ( + ExportMetricsServiceRequest, +) +from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( + ExportTraceServiceRequest, +) + +here = Path(__file__).resolve().parent +report_path = here / "client-real-relay-result.json" +report = json.loads(report_path.read_text()) +records = json.loads((here / "runtime/browser-real-relay.json").read_text()) +models = { + "/v1/traces": (ExportTraceServiceRequest, "resource_spans"), + "/v1/logs": (ExportLogsServiceRequest, "resource_logs"), + "/v1/metrics": (ExportMetricsServiceRequest, "resource_metrics"), +} +matched = {key: 0 for key in models} +trace_ids = set() +span_ids = {} +execution_links = [] +events = set() +event_contexts = {} +metric_names = set() +for record in records: + definition = models.get(record["path"]) + if definition is None: + continue + model, field = definition + payload = model.FromString(bytes.fromhex(record["body_hex"])) + for resource in getattr(payload, field): + attributes = { + item.key: item.value.string_value for item in resource.resource.attributes + } + if attributes.get("service.version") != report["version"]: + continue + assert attributes["service.name"] == "inkcre.client-web" + assert attributes["inkcre.peer.id"] == "00000000-0000-4000-8000-000000000001" + assert attributes["inkcre.deployment.id"] + assert attributes["service.instance.id"] + assert record["server_auth_ok"], ( + "upstream must receive only server-managed ingest authentication" + ) + matched[record["path"]] += 1 + if field == "resource_spans": + for scope in resource.scope_spans: + for span in scope.spans: + trace_ids.add(span.trace_id.hex()) + span_ids[span.name] = (span.trace_id.hex(), span.span_id.hex()) + if span.name == "job.execute": + assert not span.parent_span_id + execution_links.extend( + (link.trace_id.hex(), link.span_id.hex()) for link in span.links + ) + assert all(link.flags & 1 == 1 for link in span.links) + elif field == "resource_logs": + for scope in resource.scope_logs: + for log in scope.log_records: + events.add(log.body.string_value) + if log.body.string_value in {"job.submitted", "job.started"}: + event_contexts[log.body.string_value] = (log.trace_id.hex(), log.span_id.hex()) + assert any( + item.key == "inkcre.job.id" and item.value.int_value == 900001 + for item in log.attributes + ) + elif field == "resource_metrics": + for scope in resource.scope_metrics: + for metric in scope.metrics: + metric_names.add(metric.name) + if metric.name == "inkcre.operation.duration": + assert metric.unit == "s" + for point in metric.histogram.data_points: + assert {item.key for item in point.attributes} == { + "inkcre.operation", + "inkcre.outcome", + } +assert all(matched.values()), matched +assert span_ids["job.submit"] == (report["traceId"], report["spanId"]) +assert span_ids["job.execute"] == (report["executionTraceId"], report["executionSpanId"]) +assert report["executionTraceId"] != report["traceId"] +assert execution_links == [(report["traceId"], report["spanId"])] +assert event_contexts["job.submitted"] == (report["traceId"], report["spanId"]) +assert event_contexts["job.started"] == ( + report["executionTraceId"], + report["executionSpanId"], +) +assert "inkcre.operation.duration" in metric_names +report.update( + upstreamVerified=True, + upstreamBrowserResources=matched, + upstreamServerAuthentication=True, + relayEncoding="OTLP/protobuf → OTLP/protobuf", + exactTraceSpanIdsAndLinkPreserved=True, + correlatedJobEventsAndBoundedMetricLabels=True, +) +report_path.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n") +print( + json.dumps( + { + key: report[key] + for key in ( + "upstreamVerified", + "upstreamBrowserResources", + "upstreamServerAuthentication", + "relayEncoding", + "exactTraceSpanIdsAndLinkPreserved", + ) + }, + ensure_ascii=False, + ) +) diff --git a/tasks/observability-foundation/experiments/client-real-relay.mjs b/tasks/observability-foundation/experiments/client-real-relay.mjs new file mode 100644 index 0000000..e9ccc92 --- /dev/null +++ b/tasks/observability-foundation/experiments/client-real-relay.mjs @@ -0,0 +1,72 @@ +import assert from 'node:assert/strict' +import { readFile, writeFile } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { fileURLToPath } from 'node:url' +const client = fileURLToPath(new URL('../../../../client-web', import.meta.url)) +const require = createRequire(`${client}/package.json`) +const { chromium } = require('@playwright/test') +const { createServer } = await import(`${client}/apps/client-web/node_modules/vite/dist/node/index.js`) +const credential = JSON.parse(await readFile(new URL('./runtime/database-credential.json', import.meta.url))) +const relay = 'http://127.0.0.1:35501/telemetry' +const version = 'browser-real-relay-protobuf-v1' +const vite = await createServer({ configFile: false, root: `${client}/packages/core`, server: { host: '127.0.0.1', port: 0, fs: { allow: [client] } }, optimizeDeps: { entries: [], include: ['vue', 'pinia', 'zod', 'zod-class', 'zod-config', 'jose', '@supabase/postgrest-js', '@opentelemetry/api', '@opentelemetry/core', '@opentelemetry/resources', '@opentelemetry/sdk-trace-web', '@opentelemetry/exporter-trace-otlp-proto', '@opentelemetry/sdk-logs', '@opentelemetry/exporter-logs-otlp-proto', '@opentelemetry/sdk-metrics', '@opentelemetry/exporter-metrics-otlp-proto'] } }) +vite.middlewares.use('/probe', (_req, res) => { res.setHeader('Content-Type', 'text/html'); res.end('InKCre real relay synthetic acceptance') }) +await vite.listen() +const browser = await chromium.launch({ headless: true }) +try { + const page = await browser.newPage() + const accepted = [], errors = [] + page.on('pageerror', error => errors.push(error.message)) + page.on('console', message => { if (message.text().startsWith('[Telemetry]')) console.log(message.text()) }) + page.on('response', response => { + if (response.url().startsWith(`${relay}/v1/`) && response.status() === 200) { + const request = response.request() + const body = request.postDataBuffer() + assert(!body.includes('PRIVATE_CONTENT_SENTINEL')) + accepted.push({ signal: response.url().split('/').at(-1), status: response.status(), responseType: response.headers()['content-type'], inputType: request.headers()['content-type'], authenticated: Boolean(request.headers().authorization) }) + } + }) + await page.goto(`http://127.0.0.1:${vite.httpServer.address().port}/probe`) + const result = await page.evaluate(async ({ relay, jwt, version }) => { + const t = await import('/src/obsrv/telemetry.ts') + const { MetaConfigSchema } = await import('/src/config/schema.ts') + const meta = MetaConfigSchema.parse({ INKCRE_PGREST_URL: 'http://127.0.0.1:33000', INKCRE_JWT_SECRET: jwt, INKCRE_PEER_ID: '00000000-0000-4000-8000-000000000001', telemetry_enabled: true, telemetry_peer_relay_url: relay }) + const withoutJwt = await fetch(`${relay}/v1/traces`, { method: 'POST', headers: { 'Content-Type': 'application/x-protobuf' }, body: '' }) + const invalidJwt = await fetch(`${relay}/v1/traces`, { method: 'POST', headers: { 'Content-Type': 'application/x-protobuf', Authorization: 'Bearer invalid-synthetic' }, body: '' }) + const { signDatabaseToken } = await import('/src/auth/index.ts') + const projectionResponse = await fetch('http://127.0.0.1:33000/configs?key=eq.inkcre.observability&select=value', { headers: { Authorization: `Bearer ${await signDatabaseToken(jwt)}`, 'Accept-Profile': 'inkcre' } }) + const projection = (await projectionResponse.json())[0].value + const projectionResult = t.ObservabilityConfigSchema.safeParse(projection) + if (!projectionResult.success) throw new Error(JSON.stringify(projectionResult.error.issues.map(issue => ({ path: issue.path, code: issue.code, expected: issue.expected })))) + const unsupportedJson = await fetch(`${relay}/v1/traces`, { method: 'POST', headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${await signDatabaseToken(jwt)}` }, body: '{}' }) + await t.initializeTelemetry(meta, version) + const carrier = await t.observeOperation('job.submit', async active => { + await new Promise(resolve => setTimeout(resolve, 5)) + t.emitJobEvent('job.submitted', 900001, active) + return t.captureSubmission(active) + }) + const execution = await t.observeOperation('job.execute', async active => { + await new Promise(resolve => setTimeout(resolve, 5)) + t.emitJobEvent('job.started', 900001, active) + return t.captureSubmission(active) + }, t.submissionLinks(carrier), t.ROOT_CONTEXT) + await t.flushTelemetry() + await t.shutdownTelemetry() + return { withoutJwt: withoutJwt.status, invalidJwt: invalidJwt.status, traceparent: carrier.submission_traceparent, executionTraceparent: execution.submission_traceparent, unsupportedJson: unsupportedJson.status } + }, { relay, jwt: credential.jwt, version }) + assert.equal(errors.length, 0, errors.join('\n')) + assert.equal(result.withoutJwt, 401) + assert.equal(result.invalidJwt, 401) + assert.equal(result.traceparent.length, 55) + assert.equal(result.unsupportedJson, 415) + for (const signal of ['traces', 'logs', 'metrics']) { + const response = accepted.find(item => item.signal === signal) + assert(response, `${signal} accepted by real relay`) + assert(response.authenticated) + assert(response.inputType.startsWith('application/x-protobuf')) + assert(response.responseType.startsWith('application/x-protobuf')) + } + const report = { passed: true, browser: await browser.version(), relay, version, jwtMissingStatus: result.withoutJwt, jwtInvalidStatus: result.invalidJwt, accepted, traceId: result.traceparent.split('-')[1], spanId: result.traceparent.split('-')[2], executionTraceId: result.executionTraceparent.split('-')[1], executionSpanId: result.executionTraceparent.split('-')[2], unsupportedJsonStatus: result.unsupportedJson, sharedConfigMutated: false, cloudExportUsed: false } + await writeFile(new URL('./client-real-relay-result.json', import.meta.url), JSON.stringify(report, null, 2) + '\n') + console.log(JSON.stringify(report)) +} finally { await browser.close(); await vite.close() } diff --git a/tasks/observability-foundation/experiments/client-settings.png b/tasks/observability-foundation/experiments/client-settings.png new file mode 100644 index 0000000000000000000000000000000000000000..fbc99534e1e214cbe0470a65a15ac98ce69151ad GIT binary patch literal 59502 zcmce;byQVtyFQA6fg%PNDA*t%q7n)ciXwtEEK)*LQb45}15pe_y1N!FAkwH}(RrAih#OShxNx@wJd{OCj-5c<^(XVPi>T)*S>a*F)fM|~#BVumzCyX5_@>ue$aNKaQT?+(~_qCwI2pcBOr! zHctKxJ$J(9zr}3HfxM>CS6y~w7&d&g)yT0(bXZ>4yyKA7%6y;8_5<3p1rE#E8gCgc z%?#GXTU6Zr^*&BFOxXD}@v71V0T)cZyV8}|6e%dZIxLi5pnKw|mGmtNhZhH@%Y!bJ zj}M60lBb;=HXa;DXY6rA(>Xz!I^)cT>J#yF5;XE`+Oo~5w)maD$0`^3^U_%EwWobS}!f!%`$yg$Y&%8NZf3UwL)8zT~1H&tt zc7L743W6?KCgJhd#;%U#)DQkruf&@d*4!kex(P1)3>-P=!tc3FaCK>N6E*$D!(7+h z812V9KeLj^r&>4lXPNd%mFzib-1=$PC5!vpx_165*l>T$E}x+*)36~)Du^#G;^FV_ zbgT*~iyWHycI~;=gH~gCvx5!EV~4(F=@JEIHE1_or&JvNxBHuXRs6M=yT#t0kX>Ln zfA=Sr#b##M?}ABt=TIT*2>Ep&ho+#e#P~>@L|AJ)ixAzw1<|*M>Yuix>AKM!J;p_` zc}vLAm2{J$rS`Gup@uSUFFc@lXPc}?J;jM?nFcO3`3IS$0(bt%wwr$xub5aX8!Hp` z(Wr&5a)M=kaPUyZ_0Xd;1D; zK}$o)1v4WpF{7&5ZJ5k;I}^K`HSxRWl6Y?v!}5H){&5E}1SSCa9#XEDR+d?JYf{ zp4~6*CTG05vV7&y4q)7ms8 zsb5;`LA@$Sui(u`1gU;?IIh|!-O5@k*!)a|QX$DcRX#L6B|rubkz*-OVL&J`u@MOI0Rj*bwqS{%tF z-RAjv>S=6mu2IX!{;G)hpSM!AzdgAAdP{uIb84ZdQ>zznzt}{tDPP@4WWjBsmV@K>7ujIvoQFzA)>+m_u_v{OLbE%`auWNX+V0Br# zZ)D~srBYpm!|Jl7NE7x4tGw^ZL4I+cGxoo~Q+EVY75;ELth9Pe#MPG~;d1<+M_pei z_lRw$3!yV6smrtS*nfX>up!UZ(v~9XmV6GjZ-zl_bez!gR851XqY=#{R)vOvZ>q(l zAn{h#2Rboa|r4CZmfSeGe!8h>a}J zEZmVI?)D3lRN%$r+zHpgB=MJq^$BsPJ8U~i*@lwjf)#^m6IA<6&r=TvYgs6%oux@1 zqb2<0@@&k5DjX+%F|$s7G^i7&`}B!1Utl@o+JKV#+*2CC8}CoNd_UqQv=aL>bg4Su z-Zn4i?yub*4jnc_?A&r?8tHoOUlL8JTq~KD9cG$z>3gwbTo2y)@phy+Bi{Sm2SKAD zu{a8$qeA-A%d;H;Q*7C-;S4v9))?;ha5#w)$d~_Q^*4NAoVBq%oF29HT}ch5@qchvxb;wRZXKxFe z?>i-PSzj?rHG^+*)c4$-so5SMO~oQ^A+duxeX$Bc)h)3 zJ{ZZGzPl{-&W{iGN_3om?OEBG5L@rjmH%|w*KWSYr~F6qw^#cr%wjhU9g=9cuW;%k zzmmAcNOKSOuCxNx;6?Am&bvR~g<&4elwf@@u}bZvuCX+j>4nVIeJ;I`6HQEqBjK_B zWWFX8fI&l&(XponI)n02;&gTC5*8O9{rOoJuT&QnJoADop6b_NW4?V3oiUj%V9L-} zO`hhk;C<_S+xx+1YSAzjH>-Rb5>t)Vj`_~Uu9;&Y*;W-HI@Xld*ZKMB*Fc95JGYqi zOV`@w@1tM(Pu@${tCILJ(vq3M*0?MmdE=Z}ehtZX$qA9tA?;?qA&OK& zrX7xGYeAZ~Gn-f#@JKIjl&AXHEsJK#W7K>~LMlWs&$O?+uOj5a1}g#Yrj4B^#NI1+ z?LD%k(Ik#`c-niiSCP6tSyPB^1l!~Tnqm*5{lb)fdu~?4SnhSET?|ek@5UCMYbL5@ z-~}Tm#=5_~a-lvr^Lg`O-xYFC{?bIrWM9SV;#fiQt*qTEA%pCy>HTwme^1L=cRBA1 zEXdqQtn-sNE8Ds8E(yQ$*2~-9f4RH7Fiqz~vJo&J^qd#jPnW7w9;h#JvF;CEJjQ#t zEhqCq*B6Ubbc+_D-*<~%?22&x(Np5nkgS>9nhNNGy4uU3ajQ;??YwIx#ZsuHpyg0K zd(nz|jzvZ=g~4|WHOsN~wgSi1^^{bNIX{UWMUlW@Jk?*tuJ$qgOAPi~^qsCR&S)6f zYcS=UEX8PhUe5cGaFJJmnjYyZi@d&gqbLc#YP(BHA_*vD7WS5}hb}(Qst6Xq4tvyF z4FIxOeyS+=ido3o?V*JUi;+75i>d&+Tm2@~Pj?DLkG*I*^y?Rb_IO(P!4VRBe!SYO-Wd~^? zkreNp)h2wqWIm=?{%T{>yTUQw}6|5%!CMSGTMjCe9r?z6_pD=gEo3iVm0x;0Wc zF)Kh1)wv0nRt8_F#0QgYl7(}$bN60*d3F99RXp?3QDyh6^>&LhBaW+gX44m2bFFKp zQbRO*rkC$7MkzV=c@(i%JZ%$P8oyfNk+eEnu<94=`pJV?IthRv+vkjWbBuJzSi!2s zfD%Vg2NjRUgnuZ9mp_5NgIVnRE5lF@{!?mba8btp0IfKsn2+-(ZI=G_=*Cx6}`NvYSgcZ zHea#Yjg@kG5oy!X_0@$#BRBSoNwFu(hV5}RbC)bfBgH(JiiXM8q89>pCr`RpVG_0D z>MMdS`S6-6Vb(5frQWoi3qL$4jlb&L)4V$*hIILj9xb9`A2k}6Szd68peL(5W z-v#R1-ySXgxOH6MAHH{ub$YNa0R44kd2wXy#6j00s_)*fb~zFB@raz5@7b=`2RI$T zl>khPnm&j)cL+OE(VbS7`Fs)M+xwM3RmZv=-}kIAXc3oWl{?x|FgATxvmt!_8m)9c zFwn5q&vDaM;-?e`*Zmi(b6eW0mmjgl^{jTwtRsdy#i#ZE`su$o-_EtGph&vBHvRu2 zzi>^nz#)%l85EcA?(4xb*h_p*;m(kG`9DsN!N8?+)ARnN1rYE1pJe0z3xwZ_f888F zeWppr$72~l2^|HF9fagT!G8e!ZLB3T`KHgA3ZYfIkN2BC=vJ)Ew3y5e)`3c@D@JmH zvMkL{7EG33T3dt8Upr5JA$y{WtW4ENU$rwVbXtcx&IndLF8Y@LLFZ*O1eCYI!uCTK zQ=l9S%H+B+T>7}SmR7CdYP+=S6I7U%$unv>79$`HkvH6hyB<`AUI7s^Y|EYoF-M!! z7YI@qQ_nCUJzh^aOwkHP)1yK6Sge4!=<#PSUQmq#&1qV7Imas{M@t2Pur`6frgmi8 z&Kje!eRn&Yoay;2Lewqv^2&ocGU?{JIUrJN8%}=W6|ULr+I4Z?Y30xU<1@m%C3s`q%{k?1&fF zbX<-a1>yBMts-~NppPQ^dZ-vnH(87rO18ybY%xs(?9MZIao{PXwgNyGOeic+d+3!7 z+-h+vKxg{ZkuL?@-yFO^j$>0wdcBAFZLb_)HTC_9lXIKaw#t_y$5;!g9s~4nsC^s; z>AQNUbG+-TQqnE7JxL6&=?VNLmXNf`Vn@WOi~7|kJRZ|=Uoz_tJwk4t$hTh%d@AX8 zzNbkygd`fkV?_I&jYQ@$?Um@G`1Ssz+NTGACe~ScRpDrvwi%*V^6cg(W=b$JEkLm~ zz7l01?!}5{wg{9Mv|Ogm3bwbhXb`5IIM9KT!~kkHJT@3nWSIT*{+bANa4mP)1w zveuA^6WUyFSwKT%2T)<$6Y^AiQM_I3!XUvXc78(hqY+vfM{Un21AGagR-po1u$>u3 zHS?Pf)($ph7%KI7(lL1nd)?42^{q-!;Vh9?W8^c7wYCJhVOp+FRI?MI=x%{aF)~XX;&hTrJ!S}yQm3Axb%j!gASM0;N56ldz}ud|W{lGgr4B}vAEL)D4DJM!4iry;NA zIaB!dN@2FhN2vz6czm*FGA79lXQ@gF`KmV8b~dT4#(~M({oI}F z%XtrfKVJ#vHT`Z^_6Z72^fSZQIPbn$Tap-$wHl`@9hL*8TO}!&}c0-7uy{aSKy}Sj;51Nh~w?1jjzNF`#nlN|j zAbnu=Vv?lq)AFyvKJiq7z-v7}ac}zJ_upqfnGL*OSrGijioIXHPT1(5=gNQVn6Vd! zrUTKFyifDm&fK5)^DD2N^mztk4i&2~*;@vqyA3dWMj4CSGt zvZpUF5NcJn*?{n&?@RUW;fu?Qvk})`YCFHQUT8oy=Peu-a%eRJjQaXdkJsMymqBZC zx(%%o{k?jAzb+8Mf!=nV$w?9CZTmj*3o#5Y=?r_ZDS3vlSNJ?mbUAVT_5NOKRW`+h z2Qy0d=CGF(Hk_`Q%#kRt(nx1>`&4u>4j@6aEVQZkzow|3Wi(1=^I1Bg#q0${G zV4xDPVLXrVIe#ryv+2)4eNRXg1_@%?ey-)k`0gc_U{?dib z8?)}T^ukDG*jKFl?{BY^Vr7mFb;*8in&1y?;-|ie-rR_iLBEy=yeP{7MZ@U1tl0&j z-blG|bWH42yv+;-RzNChKC(bVNbhdVwNjrDbS?v-a`E_ZY==Rs<>>}?;u;Gh_Fd_? zXw;rN4`2wv_I%q_SsqD$E+#qE;N1@~5+p1?KcQTl9ZRO=D?EVjLLp}R0V-N}YCREWTqhO#; zj0KC~M%hGF?nx_M<~moyI&#!YPj)p$pA8`BI!toF-F=f$Ykq;7)JDqA{pYl&(4w%X zz^P?NA*+zZ3S2#*r_0SZ%u}}z@R=hl6LvWbQ<;dUtvJsT_A+4u)W zLMh^}Lf21272|e#S*de<=1%tR6F1J2OHd#@8R^d3hs+F{P4-Gs2MXHTII7GZ<#SxgCFekhm|6ICJO;Cq z#gh6HItTf-YFHUwCZ5t+=1actsTs3kB?2JeMmd&#O0E0gu; z9w*ww@l5b&twIV%`Fbu4)2oY7-szABpyj(A;EZli`|hLZ;D@1wYR?3q6Ln+c7O3m{ z?ItAk{)E_AdtRMFN7nWO=TAUxnkm6H!e8a!+RJaJ_d=3hgwvy^;xN~>ImZ#LF5?;@ zVPu!7?dH(nUn3gFlF@F(omk|f?>DE%J&6)-o*Jkj_D71YKq%a?ES<~R_H&F2pIUxZm1>0}*5o zlLSOfZA{oQHAoJp6k^{AMD7)LT$~y(#We$C#fWsasqOK*}us4jK#8gLAm7T}2^(FkdSg;v53YI{!V+WlIGbL>2K}>JZOmc3VrF1Yq@BPNJW%gjzRH>Ml?8+&j z!U<1Cu^!y}ysneBkzC$C05Yaktjj}#goR-TahZ~e_Us#oL#`8Hj}{?1L45sk#D`Di z!(AQsx?BvF_)3Fbt{6^heU^pb`%mVcISY=^D&C}B60z%5PXDmlt%Ga=&K5$LHyPeFgm8Z z2Sccm<|q4z?lI9`_;`cYGm?xX@XC|)O^R<6?$tp)BU3el1UT!SG+5N>1%+Wtz-%Y| zQW>zsB*7$=ttaIw^vHXWN=%=ouCShMndOa_E)V3b3!91$jLW*zaw|dk+_Hexcqko{ z_yCiB%joKl+}H4y_8I}T15#9dQ4M~#1c#5cLe0%7Jl~+xKVprvB<?TR!n1@;v5^TB_wx_L zXHOX>=i!mFl38iST2Ddk*IWgE!WrQ6yiYRA z;M9Q9cmd*RkuM~C>PaAfmSv0(0*eq_u7QhAp4+!NSs;m%q7N{`%=ff_TOnfVbz?|1rQUHlKToh6;lBelG4a6AQ1?5;ljrv)X{XVpV`8w`28{LO|NVZ@ zByZY#*;_b2jz=w6{L_HUMFPx{yA3KJaC*+rg_XJ5C_5}qD=l+N8q`F;Xo;k1t z;Tg0k?MZ08&9c>d>w#Nep}xvB-C|QTtZQNvB@<<9#$vV%DCPu12TVc#+2APP&pF zEXv@9@r^yWJND%dZ|0M%-T|Y-O=&}8=qS?-{MptJ`_LX!_y=ljF8llR3R;fdlDC@? zb?!ia!CN+rPKNfVBq}Gm>|vH_v{$(NaYgM=49Y3qa~XSL@{cZY;Yqv0I@Zm|pf)+L={PkjtTDqGf7DIhm!1FX-;gYg?1(LxF9qcA=pHqLL10+jBrmy*W0m#R$} zmmjZ(a*)bTr@m;>gWZ~-mRTNr6Oe2D-8)`B;3NWY_s})`1YMVBNEEVns$-;+S^uKn z9E6GVjUQ+XJgRo?NJ+ta4AXCO=vJ5nM~s@&(U1>v>PUbr+lSC6HX}@6p3$7~GyBAC z8-v8=A(?%rG1~4B2<7shiHjRjt9br3xR!(SCXd@|z(^gn0{AbV zr_@HwjBZ0X!$iTnX^B_j5UC-Q`()ORxAfesMOo=49U2ka8TlJ0mMoH}3bqQ=h6Q6k zWgJMrgr*9KRwH#2Bvd(78&Gb>EbydCooZQ4Bz49Tq>I^1U#e8rY}R(eE2VC;XU0=D z&rOtYkmBk)5VdIXAPp$9TS~ziKecXDt$uf$R(qz2=Hg#f@fH6kllmGKvsl^WIzhO; zlh@82IH&zgKp8t})As#Z|J>;A9l@DB7n;NQ)|1h$zwQ-#rZT>P zZp9+REa_kS5B&)WuBW(f9W0Xpo(pk@Njf+el>&bEfv8!E@!u@}%qfbEZik<)b^QNt z#Z%a9XxC6P{U6jxzuehTxNYAV zezX2}bg9t1Fn!L~EJHyl_vaph3bLlq9g(vwU1x8aqKL7425aI_JEXpw1i5U z&YJ<^Nkdi%7j-jUyTD@^%u{>&VP3i&yindzGX4oD4>twH$V7wHf$;m*p5!$Pw}J_4 zgeu?}0&ksy+z#%+s+`jKV%Kp6-#p7P4UkuCWxJ)hhkYeLHK3alWfwX?ni5&R0%nlG zX&^;#5mME5kP{TZV7(bZqo9<*y^yc_4c?1^d1DzpHC(Lcm&cCE3PlyX)KcNBEgntvOLx?Ai?W zBG|ci3~j>hhp4{N7pV2LF>8zGw}~>m{AUJl;F=J996ijtkEd7bJL|1aDFZT3$~>4W zpe33s78?R{+0y6HYPk28l3ZUyU!-U^I-0*B#b`)ZoC@wdxJ?T${dlqQ* zvaqC1K9?+p&oa!8btt_p1I5iIY!OTu-~j^k>y`(iPGKY91t4-ThX^Q+nfzT$;?&qun&gW{n}Xf z^C{X!(ChMowWB`8e?~3eo0Wa2{C5s_#JbCJq?sE?b+8HQ@P&4(2n_J@+v_bM(z4mD zwyvYFdd;|6afFfxI8bn$gqxOs=bPK%j(9h*l-_NHy)9o@X*=NX%~NXW7Ed?)b5S+| zRqGlZscEQqw8^%@s-T4l?odu-g1zr3%|W zPB9|SHfhCx5htr=3_?SIWddIRTF!7`YT&~CMy>TJA5xDQV*n3#A;(Dz%mUZb`CwWSjKNcw0z~1h8(yKRyI|Z(qzJLW6zmk$Edh6_ z+6nssI}|wz^lL#tDlkwYaD$CnA-FgU%|l{GLq7&O@#NP^sc_F&s{LkVI;~|=ZaTlS z<~Q|kH|ZQC?X4gxh=+ZaX2%?G*}*`MJ&v1Ym)@i(M{ve$(bxOKBnZ_(fvlik=FiQ# zEkO494u)*2zukxi81jW;kwzK8ZX`o0R}fD(@oR8x~lY-5$+>cY#OPS7rv_c9& zMlS&+hMB;+!=-DUM3qFx&*hHIacYW*nC>-EH+A{mBZzPth*H~uRm@zy@{u)1($FWhari z1Z+g+k;-8&`j80>6AbOL<ESeq)t;t;?^-rI-7j`;DDR-U{tPI)Y6>Em%{w+K#&n7=rCY z{L&?(W-iJJh;=N_bm8YC_Jb!~eHK%SAah9l{^9P~r)l>P6-%$q4%y@nPztRt!CQFA zmjU%14GE_)ooKcc(z&9HPIUKP(XsK=T;=d$Ec@z5jfHU<8S7?MX8mm3p4&7lbtkXK z*7QRn(h-pSz7iJJ&2Pe*dkHb#b`prmE!yIe2N8%#{yp(W7k&0Ni2^hZ(`fL`2 z%^U*BA`MlG4huiu*>TNrx_o!Scf)v7nl52Dqv=Zcp8Y(^OVnE%%%9PgOpZb=Aq2Pp zF`?`l_5DHjV*6nXaEeXXb$HDY+v&&OOL0iZVJTL2wSXrQYT&3J^!pn3(<&b*o<@mz zTzB+1AqKyNwuTv|YxHsS_y6-xuUXepR$xwGLtQGQ01UolGo^=t3Ul6gAyy2bG>Dg3 zK?+w74P%S~S(YMc*qtiS2-OcvgP5H;(pH3*oiDuHw}Yb4gC%?V)5+-vGqTh?NVkXW z*b9V;j%G{kjf_>YVAp|vE?R$_R*zUhC$|4KB>2Biv|P(%%)?Et9eSYPSCmEG#$>HK z?reIvjf=R9|J5TN|6WG_jZ-A!tCNV=W8vA0GA%}1;|OW#j=*|1vu?Ma008i+3HcL4 zx5>teuogfqc6RtfBEzUpa5=|9++z|Qk=0S$NDl}AXgxF5F}A3PxrV-tW}2rJJqd+U zDww}1UKgVq4fD3M7-F1&NOd2L)H*UTr<-#uO-|N<4=R7=MMFSIfba$)XJ2^r`R(r8 z5w+PMJScwy0ktlS5g*75z!6vChY~Z}KYTF&z*qoOlb!s4ZVJr@O^Mu~Wsuc;2wpq- zn%d{T7vh*Z+>eVQS8h~LE9b)^--|%?PspIegg;`3h{}zD3xMD2j*UY5=ax8oKsoG& zBR1N|u)b*A4`X0m9Ht2NjsiGD!Wg6u`)+36#Nii%sj5kk?XSPw{f){6uEc;z>^W}S zCHnYlc+EZQXDH*O1?T^Ox#FgX$7$ZvO~leyy$JYO*n>1q{LI0Ntq^?)`AhME!l6m3 z!hBk0;qCe>3qtd`sw*;k|2$?2ze$4#g$YgzFzV%Z-PZ-FPDr6TJLK}|;4>q+mG9M2 zkNM08V}FBX)%QF{MB)MTDdOx4ouKrXF%8~}pLW**L+8#hyB%N8K7O z0$`F*(XlRpKh5NTE;Zx~Y64%HSTdD^r^2&B_{_!Ia6SlKJX44K(z3tubBFfxi+>>j=<&H;HXs2K2Rp7< zlPJ*!;*Mq<;mDushNC#b^6X*+LOx0f%CZcI7S%?Aao&yQz5QjQ`n5`~x?apyFsi!X zav+k7@P+B2zWeP+xpR99Fp!4p@_;hAC59xahic<$&`C-Qn>W&5k2qBl7X-?e5o03{!dK>{Y=jlHJdE@nY4iT%=)D(=Z;>Rqmk>gt+lACjvR3qS@Srn4UmBXRi-3!` z9Tq}=o`m0gYOW4>PL-S6cEfapSSgoUy!}A;Yn&3&%1;PjmS*H;LHJt7pOqD;Ld8Q) zF1=NcXGs&_{>7;gv z{CLA@+Fxe(x`?w4V5kB?=m2_CcNqsO={c8)b;h_r-|0<;tk0 zl{dd~<&J)YUJ7IdUmvP@*HKr0e^2_*SCr%?A|#$YR}IUr#t8_`a-jMo2$#P0J5;gbja`GUTD+Eg960LMo9Uc7xAmK+pKIVyO$JH6IltAn2X~}U^M26v>PN^ za|19rNj7+HNl${WOZ#)h?Y0Hc;Nq@hq)Vc;LHw=LH-J%bdvT&su6Bx;#;$|uFm*`$z9SU`tijrCWJ z3hUI>W&cJeaxwYG4z|@;-)NSW3uoax$U0)P9Gp9v2X>Om-%|sO)+~@WYEOOk;hWgE z0I+P!6L(&srDIc+m{vZ?m$$9->+Pm9CL( zmHU$>9N`sL)IS_ir{~yJ-!!asacnbk2QSAc$hL0u+~OIMB8qvZWkN55>;@Jbyeq%M zQe&s=CPkLI{Ld}MUh~nJDqo+!PqAPhVa?7D)y^>}fS%I@6%lPahK&h(QofZG)OuIl zxl3ATS>f?5ZjF*jdv%gF)Ru&*X@o6

}6_9btcixAIO*-C!b5GT(JeB1j-;)U%C z@~mQE(2uMz0_ru`WV030kWt61-PHTfUUn9z{EeY`-CSx>J^2R<%ZSzT3wNrA3yTQ5 zQycsrlQ-@xcV4V;;WFZq??IH2`pb>xvUfxnJihJ6fKG+civ z)|dX|`ZfDYO$s7zxi@_-au8Cd6-*_-I9j*Nh$0zMU> zx*-J+K!`mcMlYl6j5)th+`ulj;*%Pr4(DQ7h6mie2*udUI=IYjTifI`AJ^6q5Db1u@Rs7C z(A?`I*8EuBmPCqx&6LiL`>$^OF|=3D`5i^amNy6)7JGph%ktk0=AVb-jBOiB4mEuO z!^8j~9#dwKhHl)jVMig5l%Hz~ix_nK*@ngDr~msZav|rw{Sm_np?v26?h5~ErkXkr%vrr#BKOLsps>}4$9-;Q_Hv*L0ItvrzDK9C&Z1y^j0k66 zj@$>|0751<;P!V3Q^v8Gdk#ilj6(~F+aHnAwbUCk-El#utOMaQ9vlgN#gU)F-*h=k zBudKmC$L43fNed0zH-d~nhd<(+~`odShI6@J>p#T#I;W#&vWg{@k%! zYoFD>w18Vtf_)J&c1_30dI?s=4K0X&;_qynxN@=m_b-j6s04 z60gc6Bsm}CHt2bg`lOUfW--)?(x#M3bLm$tMEXen!`)vy1Hxp;zU1j>JtOO2=jef! z7KH{{2Y{zV{l>ww>l$+ta306CvAAn6&W9MEPkg@{9>XovQ*QC7Hmr>e<2G^ro+UI# zm9i*MSx&UrX0cl$_Lw=p**T^{GLQpbx)&jez{fTmL7bhXJ;pso#Hl$SFY_Xr z*Dcw}Jkk2LRZRY%2xKwsSq#13%pnGMvySP8=QL^6f^rCcFwN2WS zIllB{=#|%ItwK}Rn_LU%4yk`iwW9c-rgr1u^poDYiR3Q{hm0LGO2k4q)JbB27azXf ze=fm=xsY-9O3-FpmmSw;Aai(R<;`kZml*@TzyD1s>YY4tkuohb#xr(^!O-vK(YdDOWQ_)_74kU;3{b)OCG^3z# zP-i*txd==Lmz49<+HoXyp$}<(oAXHT{)N7Xu3n1M?bBzi*h9lR!Q>FXvu2x^XuiO> zz?uOP5RMHwbF=IWj6?*KiL=a!qL*)?OKo?qhH+48GljtZB98ro50+5HzYwb3DDCnD zssvOjses|f*X`DBIGq4M)~})S;$nwC=UPPmu-Xpcq8F+p0WaAS0=MngN;zv{jN88q zCm-if?csMw%mT)lKbl06%FhQ=>0|JI(8szCNZI{d-zJJSiBu|S+GpS+7ZjCY^IBma zQKjnmUzrG+#?2F6M}ECW7(`}rDZJrdHaCLPDPL1)EPlVL8TAqA*rDX;K)`f!y)&+R z0s4`DNSydMox&o9Bao-~U($Yr>@e^m(vI*U*d}axyp+0sxFwgoEf3bXR4s4+V#~vI z_E~pLH(<>KCMW`beKjM5TgVn@Vzczu`}(0{q%uA~D?NyIh=^sn2l>qW(AXVbH^)tlafuqSs9rc0c(KOR-^597I9Ko1%{|tMJX+$# zwy*ppKc8FD0S);bEw33xCYK~9&wZ>kJVxc>`aG_JR4SjMxCuVrh4O$w37tCiwP` z=x$s0FvZ%@<5+~G_`a&E2{{XIlSWbY@%>~=!W{^`{1{I>U@VN9fsYIENC-c27h%`^ zg!jtrgJ92Im?qdD{_cL!wUyl8b1g z=EWqq7JY{@B9C8xU8y`+e~4sW+waZ1gm{D8NEQw&f#7k!LLrg`dBLOs18VaRo4^AH z0IJJg0Qtd5Xd$a{Ey7od@lWI{XwQDs01 z<^)(uDPuRcr<{JgkF%)k_LvClH)|}k0Ik3pXtGR)QmSaH4p2h+ zCTBE{h=|=tMgyjK!g3-s6Gw>Lu=Jn|haiXLI8@wGJ|L14rRdodkuDizGOxC?W;=C) zSM<}D-Xdj52n3IXU6=QTV1S#f4ddyLh&=w1$Hn^wrI`KxGA-=E`VX--$>!nGfk~%_ zn=@`+x{HaEE|qRW`cd;l@1)_$#%=w{)_`kOy9OX!FA-&k!NYc%JE!Zh1noW@u*^jq?u4=8b&keIKt* z5h1O-{P>B`q1K6}kqqCN+R}B3hp#8{O~!6f7p~@$#({|g#1Srf*~8DE1mJj-(q=wW zQc0H6ir|-{#>S3n;yJNznJf4wUqex1Z2fJwrJe|3PU2;zpSBmw;S@_aq9ZCSHcXwuXIM~9)EHSl+9XJ)=Z*~^? zm-r>u7lX-@Jf@qHDSq4b+ThawAeF{ZQ;G1As#$z?JNIm5E>{^kE$Vp3s8x6@|UbKG>wTr%2GGFhf!&bGCmU7b{wn=W&7^k6xO z{^Q@zsr>T=H5dBz?-_U*Tp%JawfjrbUd;^Dh%UaQj~fZEi0<~bQ4QW0ThFTA zSW|0pj^=*N#%XWAFX--diNle>Xj}EczfIKVTX^$Sljb6ONcT^{&VIYKxMfCd-QFUu zJ4k=I^64nmX3%w9ErZIZu)l%P3xfbBFHe2ZrAB6EBhrE1 z>hg-iRnCzR8V#md7Ytq*uH$~t**mZ&ERLF<>xN$%w15B5FYluRSxKyk*3_ZuB8T6g z&)RZZfl$-Ssa-R?y43(!#osxvy7olRIJ6>zGDD#QXSE8U#U?nph&zfZC6B~XD)H4- z6&k;+4&79Fw4~!U`vZ@3ZSL&ofm?jGq7wqwo>}Qt7te|p+*a9<>h<85-}LX+seDIvO~1d(sZ#!uJ|h?-r2lx zeCpSesf%odd{=5I9Y{CKbRv8}96io1;;*aQX!}#_Urb9%J|8f_f)A(tOry1 z;z-*$xi?FSGQZ+$sLJ{3LZj4TMjv_pNxPib6*|_8TjobIuEfZbH_X+Cn=?i?Chw8l z^USP0VGs8=o^MB!RpQ0?%FZ+B+~U4IS*_yF6hGnFoSgl}{%L&qSl#{(UdD?LF89f? z{*FzG@*y+pohjbUC#f(>5?7c|ax}Mt$w$S?CsrrsJNQl~pKqv|DghrMGGQp*$bT*7yho7dtb`v^u$g@9y$+O+wtHRUX`Fo zgR}QogA&w&pS+}xeinL?^i4vb%)>l=#_S1Xbfd7z6^fGGo$QVQSLbD-r0VRT-fwe0 zl{XEa(vUiKJDpZXrQpuqy_-Ta!uCh(2zn5&82!@MMQcDoW07oN+52jtd5G&0U3;Bq zyz0}qzrE{Z$>RDyQWyhurXZ8}8QTY5y`@BUartD*B;i}h%MztHa<(~d`ar7X_DMx2 z%5xh|F%0Z}{gyZS0DUSnlI&3(>09!ZQG1Pc$(uC%y)tbbC7AdJxeo(@p(A&KpGDF* zcYAx6T+^ns9eI~s8b%S}$J#g)wW(4W&`@EEg3Z

p_{pkNg(|PnqeTwr9-nGCwQ( z-pqmPK1|XGQ=|T;p>AKH9Z8;)Z;>Ju!Fp9`6aU>DWTsbLKV=!a{ejg~+gouxgP5dK z`Wy5w8FW-ig&U7lA7!jhcU{l*i|SO-)o0PkiCUu%WFr;kYNPff@A~?&+?@AoR?E;A zmk+58o^*m`>v3?QhSq}{GNQfmDp`U-ZQnq8$j}pE?5@1)G=GmLZ18XBT>+66Zk22H zFxRr9t$=}4c#=_Ph;% zc=`1dD1hh*Dd#Yq{z*S9HK&6h_4%Ln?OMzi#X-ku0HWvA5G`PK8TWM-D?pNtk? z+b`U{)_$jCk5Q_o9ULORJ5%lVKH^M}V@=S%v;Yr%N>*zB5({tU1O7$^Ph2McC8QQnp+b~}_RE$QfBxNlDu$7M(J&*g@`DT$`ZZ5m&FR|R<(-VO&$ zmbv^);470l-fa1h?T3LtqbAn_s)^&nCO7qe)Hr?z6vM~;9g!f)gjo!qMln7+^kU{f z!SX@jUPBK~fxxSS*`KIl{Qaqndkyl+h1>0C)sx4rZ`$9cvW3c>_Z7|{#H7XSj*!|@ z=gwoKIvGDnt(kFq7OJ-{Z$4}nD*WRjP2JtW`owrs z%b-rOLOlJ8u~TFpvL|`+CTRd4sbhN9>kvnIsP>tSv8V3x#8;cYcMmBH$P?B$hwzxdwgXKDW)Pp@O(K}zqxwh+a{?no{D^_Kdmhr z&NfwK$%ud9Q)mvMjN<6?@Q~P{5LML?cD}Hy%JN)>aQYiF37Jst=?ChEQ-v=j%p1O+ z?ko)xOH=%&;aKO;p-JA>!Z+JCk`ch0T*}vwC3vk_x0G*IpjU78p@v%u<@~N4y&=d> zZh89SfD__|F|IP8hi0j_`qqJhoX_lQGW%=`?tl!lVbh1Go~#E+1;_$VEL2IE?!d7( zN$TX1aU7Fsh%3iPoOUQk z93#J;f89w1rg$GVrSkZq#F`oz>66?XbX;cP1)xt*Mrr7|`eCfp^Lke&SG&_km|mi5 z8V-3=_tL-Kt3IaKSoQm6GOw?JFV&IWx1wyHgzv=?yZ5u_Ohq8??@QUsl^mCD`%fVVY^*(Tg@!);Wm*<# zKxf&=6>&@Z^<8m%$5Y!Ya4Jpc;y&?!tG`;?jY_BY1$c|3ak_I9UA^b@_0EfX5?*%Z z4-}88PK!|rO&-{8r@XB)a`U~z{>45V$%gBYFPmgwxCu~+y(9BF zYW=>y&Ce-a^nIl@BEq~!bM!E!~9W14FhMmPCCx? zdpXu6)>ujwIXo$WlYviWGbv{2Eh@FH9fGdEmpS527BR8u z8ANuc*T3co89c}rqY#<;M7gjKi`nsccp z(ep6hnl^ysPC%YLTgT|Qc->b`PpNG~&5!?wxAzXmy6^wTsWg;SXbD$SRyHN0l7#F% zqas^oWHyYHG>|Qu(<+=~i$q_2H}B@ACN%vC$%Pf^GQPCp}0-45n2<8yH;wKq=f-XFE9ucW%owx z6H02dkGnQHCW^TDolR~Z=~^L)nCFCl+In`noA&iTKF_bfG?*(DR32QoKsmqi&6uCc zYPY<$G%bkKj$7jIE>8-?>>OTeI8RM>Ioxz!h3m`AHvlT_CJTkZ{ZY4jm3jNp10WMl zwamO5wG?3Cq;tKnH*+X%>;M8<40t59mF(QZR1xz?c!&`x8`^OS@(u z`8E^Q*f<}uCS{t8J06YCURZo;EGRN!2;m5b8(GpF&3)(KCopq)az-a@MfxX_K*O#- zS&!;$II+X09g%>EEfo6F&B_%X;rkE)^$*nMCDkX;^DHCZM_ux@sLQ$GBx){$_>l?O z!%(p+0Du|=dWe%(6f#fsyH{wE4d2}ZD4Q=DSgHJ0MbW-GG3^&|%|sQIeW@!dRc9&+ z^xqcVPsm3bIz|$tWh&M7(;L>#dDPLSom??5sPzjkwJV}#5{a0nE~h>$8zCna6x8BD zwd3Z-$gd)uUBxar3JbZ9MV6blwLvd7KbsumkuBOMqk26^VNyrQPv9DmXE?SjOO=!8 zQXx+r$*Rv)TU6$UP>i%A{sp_f_3PnGdp?G5WR1xqf!xE!+gI-@h79j*rrfxBBtcbb z{~VWW7K`;lzL!D!`N+QjSnfu!Xm@(v+Ugk8*bn6HZ7Z!(l_Tkq=o1E8oy>5jru?U$ z`8o!1b!fe*eB$LZ8o`J7y(v~Nb?BJ#c}+`E_m$`jkFPYc^|N`51nIn_&6Nu^O@26D z+$XyE3@A|O!+If?xz-jN9?Ms{#vrTz_192@P3_L~g{%iz#@pT))o=c#tGVu^kQKsl z{iI`PpPZ^Hfq%AIaF>=*@e&17VCdqd9Mdkb;v=6WSsYrqMcXGJl`S~=PW1fnGiXdY@-RWv$$J z+fwe?^3vTtdqv91R$jI2VW^U{ObC&)m3nzW&9bE`U%x3|^md~8z>4dX4#j$#I*;^y zigEt;RLXIxnfwx#1JjaY3EO45irfbB%|DZ-Qy-)piDl5BY|D|*F>n{iE&Lr27n~%& z>w8jrwkhNCloaL8;BRL{58kD?zf)K6IJPA>K&y1AX4&X_kA&$K0O!Ze$eU-#TOa zetpH)>h-PP+gi?Y@<_x^85S8M(ZZRN?rTHmj;e`wx9=evy)rBjX}^FK@H zns*GD?sBf%q_vCVZQ$c0A?DgThjZ-aeSe@+A|?5%o}(5tj#eh&2Oc|GJaBw*Kt! z_%{j9%kLt^Utih(Yyth>sd$RlD8%u%|I4Qp^h<+xTLi=oxwD0sIMDkT1brW58GDL| z>&B-rj**W%gPisc&fq}f*hpQ8&lx;6y`rA3atS@qmQzrqv+`3wC_j&IUo9|xUPkDe z^I=c{KN0>C^2A8q>ps_V7Pb-sL(q13rrSk3u+@N%-1_l(H-`X0P(Agx(m2Kk(HbiJ zEjhnvqwL4G6fhtbtqdT^u0H=DU4m)#x|km=iezG0ko9uG>0xfo}*t^h%bC3fR&emV~u^I>w+3}z@Fp&WycTR?PFsW-9k!Y&w2aUlnRdMXX5S!oQ^yXU zh-YbyM8&V>GH^mc86OBK7))KlpN-GB{rX9;CpLY|aQ8qWSqj3XzCZc+SYKIL5(|e0 zXI-wfrE6!+f3bkCg<#`%ioU~_XowJ2ZE^KQJUB7&>+#X-}HQBkFn{wv6$Ht^c$Xj@#pBc9&wXC!USY0|g6Z0Ozt|~MiUEht*rDddDntRCF%*FW?}Zqw zC!uC=v2|`_E-1FkP>|DTfKG#$VmZ?&c!s>SOBnj!=g?pSl_9i|=CMMbP;6P<>_RVT zmJKu-EYmMUdE}PV$4F$Q9n&#q@Q9aslr-9*+h?)&@J^dvyv$zcUfAAHRE0pG)?1r{Jn_o(&s&Mae`-;5(ADUd zKs(Au3N;8c1qII|g)FOorD8kxNf&lU3$Okd=Y-7!MMC`)RqS)OD4B{P zI;0{ZZUOc`LNixRr*=_o4rmymU&bQhhT|7xug;+wx@mCUS)!>6Ne(Mu0xb$^i-YzM zEqns0RR!6|U{DZ4-eOU$bCN5(Ep*Q+>h&0KzyZDS(zJs{iKZnjgFcAQYpa#hm6lO7 z(*2^DwuIAV1`P4hxEc6oj7??rlWm&f_1)WoS1MzG8h48U}>No5?+R-=0 z1>>lFWUq!KZ%(04*pp(CyftB0ivJt-fID)nqOI)m8+JUOO}LhP;Z>>v#})GedVkmp zEH^6&c5|i{21m7c44#v+^`t{z!a}|XlZ?hg6>dmwG}Sp7zNm|7tdQFydR5-24Qlg; z>`7fGUCw-)vQz)`FVyOik)_XbPFN?sqOnCLRd5e^!-5K^RQjP{3NFRRYN49P7eV5Z zgg8_@mcHKh&l-dO);{B0jcT99lzNF8hbnh#?1QYJl#U6(t7N^0VxXP7c423!A1{E@ z-Jo*h;9cs^vg>2mJ;daL+O-v>{Nxb?R-1|u zlQKLsG8mklx!M6Qn`wAg`~=IVW|M`S9(&^HPo@z*f3M^^#Qdj`l1?Um&+rey5zvVM z2I{UAlAC_eoG^%ttM>PM- zSl03&_HW{+fwMGAI?uKXh$Y}(N=tuw-c>bD_Kiz~j_ zltrkGSBjvs4^}7lSN~jWVTYD0(6sFvw2Nwdt@HwfpWs5I_q|*(Z**N_S$sd4Y58{- z5fK8-)Ne!!coTh_vJB9>3Zf}N_o;pxCm;0s@z>Jx+6k<dl>3AD+ZSi)s`0Z|K1J&;>!2GS8&n+6Ci! z=91DyhY8Dx8^LwiB>B;1%efft@#2%k&*n!!h2 z=sURe*+XyH7AKbDMqEjkc$XMBZgts&QSlN+hYznIJAvnp_D<@#%ZV?KI}!g@!p@iBjuH@V%c&ui+_LVUqear(6cQ>B=v|5?&D(Fprnp8JrRpG&Kf!) zwSZYdlU-WzMwh`SBWOCqM4FBtXkFC)1BUxU7*`NdyIGZZQ-p~>ZI%+(9hrbBno^Fz@xk5dM<q>VT z)2OZ(5GJ)@+4zJ@L?eT$KQEr}YoTbVa7MHQHbP8(t;tDJX)l65H2uxk&li0*ZjH+HE9K3sLBPbS-`{mA9da~ZQkiPLh`zNH4;?ikCm0?=)di7Tyt4i zbJ8-}m(8SQm6FD+p%%=Qifnjw2lw8>Z4^^oinVJEUBbBqbAvUSs3U_rgMVyn5Nr8{ zi`X6F+vXBN{u@_PN@8+uZ~YmYdF$_?9^~h|HH4aQEKNY3s8WZ@1wYK%Mb=F=kTlBx zxd{RdWo<#J#=wG$yEk%xh|5L!0C|>%t(KIb)Xur$hcw%=;;}*sVimZ7v;_79ZA{q_Ck0s0z#^BnsnzUMa$Y)o=UIk0SlC zEn*?O!vG9c%?7E!x)$pnDH~3G?8FTF#x$+u_ofnx9d6TVGV0rY-lnaYGfjhnCXAA# znI(?GoQjoU1o@d}f!92ZTC1kCfmq?3=LT>U`n*0nfP7(CWm(@>v@7kJy6QJli{U% zJKgST*C~~%sJ33HyO80IJyb{E>ozCvZKNK-;J(0zOi=+oG%aB zNHo?Q*ctEK!9AIuyN>XH*za@j9R=^N)0qud_8aee*<=P%Q`%K4TG|b+ke)MMc1gWW ziJM&S{THa(tC26;Sm=kjhR(>k$GoHDkdQ zyz=x1@z0Z{%i+@3e5PuwKF7(yJ>>iHH`gQf)@onc+ytF-$6apfXpGL{O#+P;TvpX4 zs;Whs(_xEgTT}OnuxK&t;0jNgnC?`IZ6a^9ss_ol__t%z%yDj+;#UOcxnDME0j5f?rlFFO+v1B$AczzP>G^7Zi&xdCVe8QKXQ z+pG;1#lLw0_wDun)9^MQ{d3?L@kJ=2H*GQ`{`r5!H~arzH~W@@0-3x&@w3~SEg_DD zqb)z$5|Y491T8XHqwR+N&1d1~A+b9Su(geKK)?!37bo!(t@nJrGY^@U>FrOrN|E9K zs>vG7aE9+N)_pZU0;L{87~u5|BYd=QV!+m~9q856g|3dUkBZ4w|Af=^q)u)Mg9daS zeMlt27K{`l2LQn%GAJIS;X=9N^apOe7EFkOeRix(7hN_JkKwugQ38+g{Id?NVs^zv zN3^ z23VLYW-`TJqVax8oMN=CKt@=B2*A2FK11kt!55^n_Tt<|uHHUF3h&BZX5@{i--gar z?KJsbQt&IaFmxf~?^3M>J_WWjUn6W(ztC(uYmK(Efix##I-pOyL1?UR1)%mPl9Es? z`5y{Nj!T(Od}_ZkZ`Rs}<{0rF0<8+I1k_M{%p_>nDj36)=-SXW5!!Nzm63R2h|^v5 zV~6-n9*aOR@(Aiw#0(6f=)>YB{0Jh{dpT3?<=7ZILm9e#FQ*beJ`K`B!qkz9j07$s zBpFvu^l958C=^)Se!jyEnh#y$4kX`X;1pbDB%&b+`C^=)O|KqJrq-Ld6NycUC*Y&0 z!Ujgj#V_3Tu9$Hnu!155ZB$B}1r(>z*%#ge*H=WoO7n;hM_3w-+ld=gmAuX=UhKzWHA);4uiz8A74x zWB8GX-~^^M8U6>7tX#8}9qJimvs|UF5fia8z!c_XA*YZP%w7X0&V}(}ESEc)-PRDw z0R|~yFGn@D1~i#d(MbCRmZGnaF>nB)AX*#4E=2Fq6KK2T>KK7lX ziDm2lrPYN<{YZ!6Qs((3XjA~^Wj#Tr68k2SWRW&QODKzdU5pkhOmJk zb=O*>9>+FWPp1%)(jEUMhd~ppoZealKXSXQ>L(OMgw??!YPv{8vFzGh0ZsP=Jp=4hI^S@BqrIdsmJ9JXU|*7i)vn0q)*rw zGXOW#W&t;)`ltEdB$UhhHO^#WdlHBSr!tqodqncxOQ|k~n%h=EgJclaP8CP;h*nM z27a_oL~gN2WTuST=o#Au#K|U!?Q=xxXN!Xc1zGlt(pM4#N3QQLt`Tg?h~hDu!464w z;jyoP1k+wG12F*{bOZPlJG})1E`D~LY$jHk2%Wlfu6mBWL?`;l7*=b>g|y|Jw|LZi zTKmw3gpQsZ8#nhZV2{0(S(raoX(D#>{~4H;`-Lcf1hJQR^tpK|JJcQ^gsWJV5QDeN zwOXnlZxa%nHrI{#UR?19kDVc!p+cQn!&;b8-{NEGNyBE{uI!|D#G~!`=g5ZFaWyKm zb1HX9fI~xAw~$Vmzr2aiQf068xTZMGl?CJIIZ+(==+?FWNbv=;AZf{Pvl5@^_?I$I zPghQ3O}avQTX`0$1jwQzTptFrb16=W{CsN=*>|5yHR*8(`c&2MAXzfdtT`re_=udy zLglirBa(Nd-yzGsBL%U~nmO0d^<(rv1H*DlbE(x?=kO;5iF)yAWZ89dt)~OwO6Jz5 zg&8XhCIPA{-=B0;e|yWQ`ReL@|8HIAC{LulY1#b3BN62`b@{;$wVzL&w(yR)d3HR` zD^r{B<|D16wOV5NAVtL1d1Sj8LfdCg%1>_9_SAz$9zqDU;mG z8aYl_o?X^G6Ow!qVuPz(FWJ*buH6g@RQ<4hvCpRVr?aQozoW6*Y0r{fSFE;6kXz8- zr6uS4k`E1|n=XUbSZm1e&Kq1s1?^v>(_Px<=f|?v-lc1K3WVewF!`gbNas~|h3gZw zX1Y~-X4K%rYX3SN&BG?nbuFEL@408X7dQ3e{z7=@AMbqzKgX^D(M3bq22{0J&6HP> zb+KJS?Sx*~9wN|O{@Hl=`a-(7O>ncgI~ttnLKMuHHwWNbK15-bX4|=6~Q$a zNFqxP<4PRSbXXzyjp2~}m-AAZ`bT-ncoOXLUDydnff`HR^!HxW3ShpCT$*OFk)kA| zZ}{GbI2E3fkaevz^{GX8$W2L$LEnzY1TTFA*Xft8JkfpIi=(}ahtiCkHxNiqKXc%> z!&-R=08*wjK`aEjCUKX+;y)xKwZ5{&&_l=x;7_$2o|WFCE%u6v7%rIM)yVWtnk27pUtR8B{GJ!cyay zlLGL~V%gR3dr9$iB6dG6XIKqO(QxClb3`SMx&ih^cO16p48vT|V9Fc9&*nFOWd5+0 zL6EtcjvBQ}LDOuDC^jbDD^IV87s1}B>5}>}Ebp*X5iWcVy6`PjX`7HwwKW-__{DL7*lg_ z5rxrfQN<1w>)iH0l{A&BxDC}^m!h&YhyVph>McH!MH}Lq4sN$yAh=nY#)>?QzL@Rd zoODabQz48U(?yT{$$({=0JpeLZs-zwX3790a1w4VCf zJ!Su?vhy?ce)uPcag}@1A}zt*$i?#MV7F4T>idWBknR77u>FZKub>kgJct=Qqc7EFqTtNrEslILMgjbQhPPFaRU*PhD(OhNBJ^gMPB%&hXJ)nr}(+n zMT94MU7mROpL9bn=}BAh|5AyP{}X@s^RwsXR1~?`*nWC2gj!Kj{LFpm*GDSyZ>Kl$p5it3q14RK1t(!7pfCfT30XfoCBQ<}OIs{qQ>x zjBt|iG9JA&4X*$F4XA&Dvg0$UO`*qg{|bc&GRO!70GrSsM*xnq;Oiv@=0PZH6(djc zp1-$gEZz@ukQz~~r~U!##9$gBV5WjH6mg%((K;`N<}YPz1}9|g1VIYo#}MtwDmoGB zN_aT`9o_Kk>p&L8Fmm6#uR7x5x&})I1exHrR{0x5a}bfO+Rl*m@;OWy0AJJM(Rqw1 z`4el7y5e9y%s69Ukf+cCXh*BOmVJ+wl=mo7n(9i2I}iB$HM~CS#$UjZ9scC6oUl9e z{(&@%qfCmj1` z{(j(*BBYLizZ+`Zy<^Z|n7(#;KXPDJUnq(zC&BAY#K*#ODxCEO^`i$&^=B#2f9fvW z{QIQ)Hz%C-gkcA5mEIQ_sQ3SRK?yi)ZtKn-6J^~cSG%@wIM;G31{5uxYHvpixEc_}a zQH-eII(A`R{SDRw1G(OcYKw0{Er92uYh(vMGvakRXj8MVqTGFr z#=W}SpA3uR{2l+`@0pv4wft53gVBcp6d0)&i7+h)EmLj+-{^B52td}$Y2R%76-waw)}(vZrv>L6Ys*gDI#pPrzyE%j|}hgiehWO`tO$- zUAVZ#&=YJ>6t;ji?MWQ3KtD41Ub|>xCXFifzX#Ae)dOA}>ZBWDr*D=BW9=>0H4s0E zLfF~vKhp!feEt5-Qcb0<) z=mzn|&}juI5Xa-Y9R)bo(AjE#>wm6(@EO*)1X7W5+6P4V{)+|NY|zDX)ckb%P@~xj%JW#EFP3ZS0v1H%Uos<;il?*_y z;6=&Dr4Ml7OQ-6v6AC;T>}>3gYhot=Umb-0-9h6kMwD+BSaG3f2(g0kAGe0Ti79JDq{Jpom@c$&TuJDYo zU-OQ5qc$8z+2_L5z|03o+uagM6l3q;iU!>4P)g88H++%o8F_c;$plwsK-xzR8o~tL zR@pz@-(2ST2mM(3TL!J*Yx0p1U)+0gy_+O(-FD^<2V$UYy%d z#7biZ4GwvYAA4984jH>PXQZ*qO5AdBowGW-vTUCHj_Kk>Cc%P{MQ1$k~ZH8 zm!FnXv-BBzzFoEIcVv7vzKeq5 z!MhiGyz-Go4=(|OqFS`dGu4+(@+IH#2S5ZNmAw)t*z9ma7XQMo>8q`V;Fo%G&0dJm zUpMevb5vz()$->>R!$Oo3IuI??5(+pS551$wkiYXgy<=yFz$J}(__QwbdR8YK0fBB=DS~PT848`V^zU0T*nFKIWud}V zgAcw@q~*KCUdlG7{t!QnYz-Uy6_L$L?!uq-^}Vx+08n5`U2k<_x}<^)*D{(6r(Bn$ z`Bz{dCx2qE2SKkkt|W$QG?vK({pDC&8w&D$bQ)ha<)#sv-&qAOuZ8w{|xFxSMo zgb9L8#ko5dWYk)fr-9&SW3a`_WpnwlqiDA|IMSsV^hLfrF*gyx_{oe@8I`!{N>;|w zy7JxI1Ir|12A99zU!=iKrdbmDOVUeu{%-vyysHn10wIgL1?*QjSAn{hVeYdNxK<+zdMxmxs)$jM{5!nB7J!U8)i4@WbAS@9XDkf`+kq%pgl*rkb@U__uJ!LVuXm)K`r}+ zOx8o!62>2yndAR-z3QS2QdGfI&ToLzm<)q;j1uKv?SCb=XN4X+_bpnPA<7Hp&(Fka z-j`UN$RXTrF!?jetwv<_|MKWO-z}Faxtd^ptxzLwD@i6SpI$BZyknSTyv0ST9aO9w z{Wpbvej>~;t@KzQ$=h%?l!a%V*{T(tJU zZSmDsv{bB)9E%p0x&A+w=urw}JRLU^bxSC# zGMn+}MXvXqY-dwP66PdY&F59jRBw7LveOg~IXe9UX`C95Tqg9E%?ZT{SsN|sxn7bU z!N1>`Ja~phT}3_rjHLQqg#$9$Qpv&LspePjq<(1cA_l&U$yINwW|sXhW<$;9UnFw9 zZH0F!%Y4SMi$te7UUqYGq{+uejV0JU_YS&hGw(|U7m~}^!$&zq%e!{=?D-OWkK$@N zxsQft`|wX{8xm(~hsD%uD~z*p(y`j=nbtzhks(Q?Gxz?Voa-s|=@)O%zM4q3Ou82_ z7H}u^s|uw|>=6J#^`kpj#Pt>oCa({bYUpDDG(Kk+Fz?d{=cA==iq0BT4^w#8`*?pp zKjU{@YF3?P460)dx#`(#o!S(c9d}BI?GZ_ZMC)z$Y=|&v!{MX(L2Pmx-f#2I(xQtU zPuROoB{7@2$oAbywTmfqq>?u4>p9r}W4iI?$WhC-K-pCG!{#(;DU9Zoxy|O@-CJUI z(MrE-R7h!fVq(72b|@uW+52FCVzS*+4R^kuHNjF;c%ACq@TeX5nGOr*8z&T~ksWoTJF(hhav87Na ztExw>oun9#E3KCeJ`>gVo!?=PnakAgmR9#cl^T0HGpbMdn@8ks4vV;`pHY4z>!RXP z8Ogd#{n{kxN=u)|&Kv~;N1x%dR8h913TR)sBXZ6uXJ;#JlGL9&*x_L6;B6H3$L!u> zLb?~DRJUewpvW1Gt}mlW;op8&N46Cg-2QTLgVjORgM8{ zexo`_|H6Z-K~uzWPZpnAO2dM3rGtn?eLDkB(18{sYgO9ULL2gr1^AYo5$U6`(-c-@ z)o4^tvF|;-%Q@KRZvEHE;lVNg`$CZ#e*6?^>y|E4oa!e)Qr@zr&3ZV(A2*rEY2=xf z6YV>={FSNZ%|5dR&8KbEngex-4=7?fcl9-iPkA?^pq$O>u+Uy1UZq#1EKPM!QZGP-wS{{_ONfZFo8D zlhem1EyX0M=G11)({`gp&!sJkyih1QU6mw``JYp%3na(hS=+l=h?s zU+-;7%4dEr$6W$C>6pF(SoNp^+Y_^5CbSe@ta_WM4EtCN9eTM^UnhT#We+&RqLML$ zh!!Y>OP22#fxC`8iaNL2LInao<>0C?%{sG0ctv;DYE#dVlnx~Irm#oKoHvhX4@-%) ze}}rSlQsPCsh1L^wjvkVhej+r+<32SPi)o;)^gvn{k4gjgz`VOi<))seq3UmOV_z6 zYVUEEzeRBRP{);vbPnq)%3c9+hZ6a+G?JRa{SSu>2@qjI}+r)h%1~G zwGO9+-#j#UCt;~~bJ-*b_(E$c=}Zv~@h;!ze;@wgMigBvF91NDo{kaTkk!Ow0T@J`sHeJCu?@08o7W}dlo zHl7Z36)qF4u?<%#G$QJhza(4@F!oRTCh@DI_EX&~y$8Stld()nA>$O57fQ4_-$bW6 zBe>twO7YNtVencw+)A~3tu=3G9nAI-E#RnY_uH#pxO*lU)!R+{StDlVH8oTH!sZ?i zO2Y+8e|$RE|g^V%&QV-MMCNtXwAq}@%umLyV+>sdXbX*qGa zL1;|R3B9!W>u|IH8d1hjz=(X-8n4lQVC!V@Og&lz@n<@-(Cds?s6pBy3{I!Aq^>NpuP96bkd#b~Wom^{_TEhT4ETNwkI zFdh7+V$q>}LPKng03znyIm)}7wYav~X-*^ahEz(Dv-s2E(>A!@o;HJFO8wwwY8(H~ z>gy|?5mN68?I2=x$ox~`ZWg5~;~x@s9(OsrG2-jd_n%CuDknS$J=6+~dB*Vy<11K8}9xiQtu^SUkOzU zuAGV&6-8@MGpEK{qq?QhETr1nW%dW*;7~{p-sTvWbuR8UR1l#*CD3%WbSAXV*C|?O zYUg|`ZqZJH!(#s2fq#HT9F$^xB@)3+0OeVFp^R+mMyiQNFwBIoRijz;d%-ubImOlV z(vZQ~N)NTby)F>g7Ye+}5r|?A(@Ol1pPU*Net2X00ft5as`WTE; zeKx59G_h3~<5+n9mbQ^_zY~euM3PyY#E>;M zp9r(?;S@&XkN-y4(Zp=jn)9)~Z9c;(ao(-ZJ^TP6Agfy`;T*xyVm>5U-*zkIy! zoTe@pXE(Ru)Gy2pQ~O0L4gw2IMelN#d++N(OCDQ{uEwW#l?$M zZxP@r>pXfK17Ukx$)8PFiMY!06**Y4QON#oIunxl3A5o1TBV|#M;(5`zRid^ zWQ4Dl-y>Mbp>$-m`Jf3;u)M>J!O^i|Rn|jrt>iI3t(U8nk?~uSh-e~n58#K+{DlM~ciRO%9ImNh9b88Hs znsL51O#U)vZnRpvom!I$p*b-*9cX}>TuV3<6cP&sZPxK86Pa{BQ6?(6AIkkSrkW)u zx0ut&9HgOxElH;3IEMs}D=WXywM%NI&nvU*q<`NnZ7;5;v{9Aa;mq$W_hlv9G#7CP|f4>%8 zXvH}i-YnR6!J0yZrRlMN;Hvb&CK&kseAg5*1G^dztC7gF6`V)pe}>Ly%kovz%2Jses52zOlkJq8=kiU^OE06UCQLX` zA-2f{Ktx*K{o4}IgK}%d@1Q41jljInE{C=G6b`9ps2q?Fb`wl&J!!VS z7P@EikbAg@(&U?^I(7*$;R#%dR=_*_9Eg!J48K*E59x-=wGU{iJvVaAMGsr%du-uyHZUESq$}C#!VjM z2;*etyW!{Wnv|3r7`w)p*9A!ggL3V3$S8jzQ%7+ktzYIjwt$gcH)bEY+q<3)WM^@o z{{lwdbLq8cU-e>1Uu$1%!|lDwq7{uiS(VCu@-()O7*RBHghD6C@4jIo?H!9-MIPJz z-=UlK=3#X-)6SG%V}5U3Hi+)*Lb7;4*Z0E)W{`ab>ICtKemq(`1s^QOlnA#XEM3#f z4|L*Gk4H5ud)&P8VG~1<8WXi!NPI;)wd&n*R!V5DG!xnNr!|=>`*MnWkrG$_CnRa5Y5ZRfzJiVtv6cU4D@&Re8P^C-aB3?KzTq`K8T&;IeVlAo(#n zUyLXVh&%Gw{3{s({h>9oA$5B%_aP)_IhYQTVpZpSSgsP7175ei7}-@Zjs zppbYlN<-oeW}>CvD+i{ap3x*Dvq|*a$f)raLymNZq2voaSdnVeD_F8z4;a(vo!Kp6 zrG#h<H&TGvefW#1{}VAQ7dWa6vYb(MTv?&HVo@ckAzVuoCeWJ!ry^T+ zHdR!Xvr|P=uS8;~P8J~I(yNSM({|sd{yT@KoZ-m#n@|68C(aZ9FEfk!|9t9oFP|3d zTN{u#e9gB*(7($(ixd?eYtHZ{&74DaxB0Y6MoGJsMvM8w{kxBv)8#zqAYXX=&$j2k zuP!-yl&rU}=`WuMc8PBA{nTDi@0e*HHFsuMWVyJ0Rd09$g_rDKk|SjkC8jg}_dg!U zoS=|c{_z6Q$fmsYZg5b+KRJNKy8UAqkw~3cG$pIL?~2M4G=s#Dp zU^P9q5Z_4z{2{a2^D5We%6QhWJXGe&R|L9FTMIMWF~0nLcMJ0f^pm`?B~UeLCs@9^ zX6atWxR2sj>6K#2UmwC(DsezBBt=-Wmr4-HuXjwAb;SuB6aF_sV)1rA2d>`Q9)hL1 zmfdzE{&6a7TYn12GhKnYKC1fCF}d-@`)GSO_WryK`vK2@`;TYR{tuhpc5M3h3*vIL z?Y7E4zGZKpFKe{(5u2618+p_EjGrbyN?N;D<#3%ULsNIEC_QqhOtk;E@XFG3j1&~X zTD*B(1-2ZsHd`pJh@R-Ba5I+Yt>1Cz^v`?SPkrv2{kECW@^$6Tv*|@sg&(x=<81iy zVK(0Tf^1u&e8TK{KCRk~hOkiia{1Svk39N4rdihI*slL*%xvcA*$?!pzo%}Oo&D8+ zzsC2)JY{keV>=4Bk2z01I17LkN6Y~|-R16S@yZ?i36dI*r6(-|7KOW(dW-3tT;+Fc z=ShwDeeiYF!GN#S3Ziw-KM)rZ?@qU}@n0pi8C6$w&i*)+-(&aTtuXWA-8P}j%7YYF zv<8k+4DVRNS6F&ar%0gWIkkK>5V8^;uA7*9x7_epV(RUHNt9!oT{S!!ckWW{?YQ=c z>xsx%3bJl*)PnELuVzo1K*Z?aCOFil`w*87wUir#qJ7r`#fFf0VfP!N8uw;-VX>ti zd;H@o_EtwqxFY^%q`Y?c&uqUI-v zTa#bq8)@~Nzj@ajfWe@S;G}#!=i^z{hJ?!yaB+qYQq6w6vI)pdab3PgSVk0;uH~B# z%n;cAS&sQrm2>7%%r~anJ+`gtnst@k)t=>%PjQ;tgj52bYjs;#sgAi9%SDim9PjOXwLG=QCp|1|jxmR3derDI4iePIlu3Ne*FoD6= zIkl<^XN;GpJeO;(X6q|p^w7#|?%UxTMS3tu0y=3prz2Q&Ug-Rs#H8F&pRW6N~9=35Kk-rYvc&_Sp-9XMgG5$K|`tSZN5}!3JW}f@+)#OR-e|3MS>=#aX zeT5Jj+s}u66J~Pck~@VvZ@jRau`IIm6KHuZcPgwTQr2aQh_`yhK?Ad$ z4d3fd4?HfK$d(g-o`GqQ)$Z3H9uN6^C|mMW;YggWb%8PSUh>L~z5Q6}oXNG~F^do< ztRA#TSAMdL_nQ4BheuUW%(0had-IPM(fVQ*YjZJwMAMXVJvz*ebguyLsMl=}Xue_= zGjTT{#~pK}G7HB@0R3<0b(+-p9_Q!&l)q@krE^9_@C>!Fg~@Q$H*5IH*{=qFCo>QHrLiet`l#xn@`BMd_O1#=-KS9nlW)I5|bnujY}5VYI@a6 zVF?zmGGcAMahfIIkX&~TEq%sBgS^nx{z(rjOE!xBGGXykcS}d&9?fRC7K)ErB}7Cp zpA)$ymC^Tfy^i29V)*|2Or{I#Z3wr>JfkkZxe3c}4VL%jIR#(tsMij&Pe2XLCb7Yu zWVZ5ZG%>L+EVD|IR%||5V$BRiu?xG)(WpF!+fA*XHzcTr8t8ny8Ew3D+;eWP%O{k3 z#zsxXZ01azwLD_6!5TVdVa6dYIIHP|nqJn4HnYl=ah1H8fBHO(&V#K z*PvRiXGbduC;v)zv!aB1n}~keMRM`$lm*)cym^PI<{4SHI~Wd(lcC}SSI9zj1!gSq zr-iL6X^;XOYp*5Lg!Zb093Br=a0wSa+HLwRZt+77K4}fuN^K8c184U()<`9fqlF)i zBUpydS3qv%+2|J&rW}gVwU>{Aq~yyaW>N!6;sONCZGB~`tJ`FF(i#pkRj2Fc88YUW z51$QLt-?L{awH-n?8Q;U+}(2W;w-22{8jm&gwgtKY^i=W-B9fXowGJ;ctO#3#!}7F zhE~t@9&xpNF1b84DxCZ%GKV(cQOj@jPAyOKPbY{{pF-y1d9~jIEIF28b_hdOh&w%= zo>MKB{BR|S^LXZs&gyB;X!pg=hf|&QcBj}5oMtTS*(;;bNt=_co%N&g{lkg4y5Y#T z&jV%lo}O`pr=gy@EynzM{xS}Uri$f@KaRT?8AOK(lu@c!aD;|yg^Ms2u4mNbdYXjN zx~>g8U95es$oql*%r?hb-cQH8j_WvpPT^1)3K0^rEpQ%=v{%^k#h)WH^uP&s<9aR+ zXU5>wOKr#Hvb)UFj3pXAW&Cyy_M)@QwU$wCLWOG;7Gg6iE z$+Q@npf!@{8 zi>hLxd3)D~HAnqQtc;WC|;)Dvgc*u@z=| zrKVp^eZwGF{Lab&sfo60;=|_+L>W<2`Qs^Z7mv~;rv(WF^QO86Ip)ld zHAz{#VN$Wb9Fv|tdF)H}*dfZsiV#`Sg4E8nio!3^? zZwTG{^7Zm1nFh|1k1a(iGoLF3u8X{XCHu&k$4dB`ZsEn^S1+9~6WUm!IZe z=BW=i3Fhy)G>|*TG`=cBT|fFI>Bn-OLj0@DW$~FB^?@W6k?16hKsa~&TG4kR}(EaQ!dnaFnREdWj zNNU2D%vIZ7qZZ;LJDQsQ+Bkb6fGsF7J6uI;;E>(U;}K_94YW6LNLUQ2T1^NIZjy<{ z8`0gYa->E)Mq7|A#bjqjhT%ch81jTir{}ENh0Edz%PAqxPDh;RK1KC$lTWeP1~+Dt zd&MNsfSu>gXg4pOc*1#8MOL{el>_zBFYE=D zg%dgV*^pZ`&h-CC&0RL~WokR!-L`mFXmNOmqc!?#R=|KcC50QyM`^0lQTHE&IdH2q zpH$4x-@Duua!NVO<55>)^71Oq%pcRhO0KqKeqvNoZM)UP%8`6!2S>BbOpMLT;00fy z8)x`mWHaiNvaco2+o*JPp;S~|l4BY`dB(=wP?NegWPh?xj51pkkj-2MhC#nyGsDGbLw(cr; zu+3R&Ql)afw~0eHwp=6! z15pV|lpq%aC|Swe3tS}%h)B-dS*O4E-ps3-_s4r*&6}F3ud1uMpqm?h=bXLQUTf`r zxUHD8NmMQSICvF-uCBBiUTPLW=nsUB*w0x79D7^%Gej~Hyw)(vqXIwB| zO-rCRCT0zuN&C;j$+n5It!eoJg7NfH^;3t2v?E(mla?(!$QF2>=6Q+lom|SSpm&Ba z*BMx7w0tg1&hFcrDD+1PPr_Bpl~%thr<1bpH}NrFYPb%)YNI_oXqIRPV3ah46`G~tF zM*Br%6f&uu2GUaWgAWt>cX0glxTTLT%IMS9e){C=%8gHInl5FI?mISeBmTJ6j6psLa0gJ6Olw)9)`H+Hs)7N~J!ursKtZ z6UKUadt%M6jI@I#Op4FBwQR>7;?{qPdqx93Y}D-VVmR>_+dlM=-=h4zyz<)-?8$M) zlhMfi=M}l%@w;pEzOT%xjXK3+3E89WUFt6}Q?b-pQeGFINOvZ;om41L%`;o5TV`=jM4aN;f@6(GI#!Q%t}4r3P8wgh zzGt$T&!ps79Vz4lH4|Ji&KHWk9Mq4HWGrOnR=3HHY4GIh+T4EA@d8)lxp`f~Th3tk zmbQ#=iN`OebxLfD)DF_ro{4&QRhs^4b9TpIXQi{A);aFR5h*)P9lF^o(QerOqbllm z-amJb(^o>UZ1%Uq>Ce~RqWiXJSIQ)$*mbT+t~5>NckXpQP7djPahFcTc@LpKTE*Rj z?GG21T@H0(^rYs8?lzz{2Bdc$O3xFbST_idw6vLen>M=Kx#e@)aT(;qqtfSk=_@89 z)5of}H1%1Z9}Vw}>aRU)*X7w(R}_SL_W8WDQ-;p+iLn^#;?To?qfS( z+`Gj2WkT`!X1UG-6~#t2@zWT-deT`mai0s9F>RBm)h)ru(EDxM?;4&8ADT?w*jOZ( z{P5_V+RV{2_Ycku%ti^@M41H!Y!MdbcYFUchw}GBMk~W8ojBD+B`|VG&aB+fI>`5o zcvWUv*FfGRC27L2TUY)*b?T?wKh^ga^`UY~DAQV~du-Z-P~K7@dSTgG+35X3C*Ah* zMEz5j((&T~8}vT>>LBkVej(r|7qpTwr^0gCNaZ^p+B9R{*-qb)|NPC?e3v}#e5Hkf zfDxaZc%}08QK34}-?W6y)r_M1X9nat*55L|FDcNWLEZ1?+a7woq?mg)ZbEYGk=O2y zO{q`2ln+>1`W$vk4}6^A_y8GE+(bz2CxwR^*E&9QT;vkg^RH33p0l>veU5MW z)rL({m7holX1`duKDu>#ja)p;TE>fc1+Gc1&~_&Kkjiw&TPx4P(g1y~i^a#2-AclD4fd!OW;nz>$Pak1 zysN!cx?`CxpEdBMtDeHODo8eD@`pDqQMJKr~4?SwCq~Hc2fs&+{N*=+0 zE>Iz7Og#;J`1)kKEkNC9+Uylr1{(|#$3xgB`I=j^&WUYK`$pIcyL}yew^?O8VL0a2 zHNI3DN}|f5YWMgB9lrO!oxG(Zyf=BZKLjP`XRp=`P0N~DM5Z?KVd`yaZc3H1@J7*> zQ;pH`A6?Nwq|Naa4uUG+mUQ=-lStDEjOWJOiBUjjta&#d#Hc-d-hp@N+O1k%^{^NF~tx$`Fm^ zk2+LNbN4rzYY!^%rL8!KXiEElWYKle=B}RWOZ4!mNzAnWWznFq%Puv(02R<){qVxW z`|aD@NK<EWBybiLDl&Tiqc9G2OIT#bZ8p z!A50E2V=5G%g_2{OzALUuJ-7DzAqTmd+LcWZ54xVk>}NJC)ptm-UqZ(L8l@vy}Yn5 z*l4a&-7@zj=Nr!29M7+QP>ltmtdK+;#cW6<+g|In5MhaC&;$HLFvf{dpaP!N#V| zbSzuW_vy9Wz9X5K(mJuGK)+qrN9j`#JJ^}4TVGYjw1Y=8SqU6ypsY-0=p2vedRuyC zl2Iiq?w|ks&d~L-0){hY3W-L_3gjTQy;w4~<`av%f9ildYiY8k%+XjlHl;K`i%-4% zg-+v9rEf1TcIPcUxy0#aoL*ZMol@oECu^Sl7i38ZU-3>jm6q0!IV$kr;FxZCA@yIm zfOdaXyHQSN&X0{x>%ZwV`!YH*rYCtq7Y{{0&hea>*!pUiwwZB`dT{W|z5O)-p0udZ9VT5~uy<77z!=<|rUd%d|##n{CmshJiqkROQ zvN|37hRYux+?pp(ukZ zjnI}U>!eEE@VBhqXl0@tlaD=`b3kD+A~fpsnR?lp83uEup!1J@{~3^HaY;40{ibfx z?I>F$C2xFaRmmh<)M3Vr-Rf1?V2uPweN(TB^$uZmZfsSRV&(}}?~S6u?cKF9JAa9O zF&BTKGDg`G@L(jyd>!m}jt&qU^~Zd- zHX&!HUE4=jI(uSeRScw+R_7HYPngZ0FPIzkW_z%#m^UeVfR$XTu4OuCKF&S1DK2$F z+wZn^zqJ3#%WTTaCr%392@|ikcO7uY+zDYR*`igtT`;~&9`1EIUmcmm6SnU#)yz@H zq)znGfZ~>wp7s5PYM#X4fS;CEjp(4Q_~!RZldmU!Z_(`QPhV3or(dAc{dDxHbnZ#u zBBM!x$IbqQYF|!upB#2xyzZHq`3Ygmx96r%E9Fn?uOStJ&Se`;ZAjkWnZD2F_l_@h zjNDN6Bf=$`UcEEsTnr?;BL8oanz!Fox^Y?R#ER=uwHj>G7-`tJyMnwFko{ z@A@vc%Gc$86tNeJDX^u-SZHV%HPLg$W{ji)97e)g94EqB4NFJY+5N^h(r(`&Jb?R}pdH1!^zU1p z?8^p|#uy3pQ_?bGDl%Rrr+R-0t#Fp2^R&vCzgm+N{x<~b`nLtm%v-G2XB~S6VjtYh zaJxO)TDj_>bjriDc%7oq%gjna+fjpy0Y{%WTs$LcAM2MP-kG`U5@uIjhp06Nn*$Ts z^SGbA<5$!z4W8*p+qSdo3uNlNX#*zvnoiQ#zBAn=S}bj|iUwkWBCC=wjbDqg$k96< zKa=#X5r3bkHL9gFDosyh#c!&GCK0MyF2$_^7e8%z)FrxM)q4H|Ufvsa@{j^p-+QCr zYP8D-f>wlnkU0?9dRK2|6`3Iyc4z7EWpeM(`iAXip{JE1hH30c+h z+Qi+H(9;-awOZXhID@^w6FBzQMDVMx9cqPewU3WSCu-{`*dJ+syEVdTPTPAf9ET;G>`q2^$zKs0*X5b!#S zepl?Zygl7VdlHww+F(Y3T2B{549QneNu-eRQf=ct$9vI*hva&-3hej`KO?^?*!-N32Gn4{~G0P+?>7#eQ> zq`RcRf5*gTy?@22uuFEgI1+PuYJx9tq`B?G>))UNFE~n3|K%;%FAY76*#r^5y{JFJ z!~${8g-p-D1}CBI76i z{kcf2`?NI>66X@pdC}tBA9Z3^fb=SPb^?7No=vQ zY7S-mJqaIqObo`%x(kt}Uc#yKV6DlXA!Bd#8{fiNfOZf~u=e*54NKh;@N9-*eF6*4(GVBi8pNc8;DA2NR`+(n$Vxzj}Lr? z_bV+sCkDYp1!(D(M5GTnI2t{Ijcn&)pixE| z+oxlmnlmVZ&|Xq#qk-fblozK4jj%wp|E&A2-M~GDYiz>lb#}}V}=k=*q7}S8zgOn}U zzQmc}2$9q0N!$vgg86q~?SZ+;5qO$eKz60qx1zx^Ogr+aLS9j*lWww%@pI5RINjVj zunU0rE6yyrxZcR>KCgAu1?+H8ZVBg=cjsd1A1olxwTF&Dzj;s@nPHB1Hg{x~xFiU1zjHE5?t>`AboCujt z+T>!1*`cOEA{fEf0JEJhqH$zzji324_=x@DiGi5mfTil0;t2eHt<(zzp62#IpaJ+T z4j?@Yq+jPY{(_323tkYLs0+U*vFh=zmAp0M!dp|@{g9u(!S4ezQYp4SHG&AyB>7amzWhTuyZtZ+vBZEc zAxb^sT;NXP8AvaQU@nTWWcR(Zyb7US;fkyk190R#f)m_SK01q8F4%%NHqkbz$e-JA z|27gx?Of|X;@1E`hJ!Qh7ZEe(?yC+9&v3-@*=cW+hnnhl+le-3XZn ze7>A!HGW+)j$M`dfu0}|N<1XQ?qt=SV;%dqMYw9nvi0YtcVOn(o}UZU0-aS@Xw4H^ zuvUqd+}i`_;f;Bx-OxuyxHj?0X3RC*Zy|d{RAz3iO$Xb}3i%!i`1d*~;(zlZHsUP$o!ot_` zEP+qB5`HUoV?UHxF}Cx%QuXYYJ~L0UMU~2zY%bqZz4rrojeh`IY6z~nj8jFt3F~k& zJYl5b*nZ%Ao*wqXX<;sTHkX_#!HrAixD(t{VgOz;%Gc-btUr<13T#g78y~ec0M{l1 zisq4c3c!`68i!4Li**F{D9neR$5@;)e9WLmFa5cHY z9v>~+U;y0Ln~e=l%tdgFRWZ@wxLhmMpuy%WS%|4Exp8yIIR5zVD=>=e?GYR&twQ~4 zd}eToG?DxZL7D*#4q!s2D6lRN>|ZQ}_CNm0d7`(8Gjhen;R}jy+ORsWHGLH7ny48Hf z{Q~`C{5xIWje@S?WnR9WaYsPNe^Qh`5?Hgrd!m*)YT+d4>{Lw#hhRbt(Bu^H7~CMR z9-EPjcRFo=y*v9PT~!&u6}vH2IK|dDW|arUls1(o*s@CSS~XWlipQaSF_aQZA}mRnl22Q;WC^VUmVE{( zj~fL{P*wTd#&Qv`s}Rm~w%D8pqN!sd)Gh%;?v%Fu07n+7&C2Da%tXpHmzD-qwb|;+ zVIw=uQBLY5WY-w_MO|0U2^{N^3T_jRuXB_LAR?y!3o?9xlyUpOO~5*nH+~$jn|N?j zW9*-xPW^gQDB%TgG`x1nb+U{Ml~)+j3;gf`57Np9&!HF-E*22&jNGdB!X;coEUdVJ zlMgWv7Qkq4QloZO@@*DSoPs^rA9#P=7XBCIt2c_~H|rYB8Cgds`^#<{LvuGtb);sz z)eQ>#a;jE5a&|NTuM@!oaSg~Ru|Le?jD@!i1R&EfWV@#b2!fJ`I)0eBGE->Oga>Y2 zbys=fy_?TY*d0Y#c)?tJDq<{snpJFX_-uu)@10w2^hA2-m)oMcW}|VxpRyuavX+u% z48e72kMG1-+!Ag)I_<5|Yg4b_b8m9T)7EAiYz)W93^%hfJdBkd9$Fd=A#qo&;mVLU zC8ob>pe7ESP>`#Of9IMJR2V8CQ^LYIZ|r!dI^C>C6v=hODBM@LsT0x|an(@$CwL*#idV`SVCxlt>7Wsq$!$*+H`C6t((XhJP3R z;{NXhylbEC5UMwP0rI4}H4w<7(RssjO~DuNzjJ-!T8S;yIpVe=$On`>a5E=VdvE2S z#^IvS>Q?nV^-~J@_axwN+(i_D(gW)4mc8Hd+su%330$arIex}pJ%4601{qpfc$}nQ zurHD;1cptLoE0e=9u&S5@!>K)Mfk^GKYxb+FK!doXd4|!Iu6#*kB%VKh;vXoILl*b zLm=pkQkY3)Yv>jlNvETyVh<(U@`w<+&j?ZO`}4QzVq16NH`Xu1E|bc%1V;w6%EjNQ z3zB87IZ4dmTG4hus{-&4-1)^{FfjD2ib;ft)Ss)T@sVS}*!%JOoS`j=NPk_DWM?{4qID{(*CqlXrss;Ly!32O=|Ma7X zxNv1BO3*)ZoD|{|_RQ^VGxZ%xzWG8kMdK63lUweK^HG~O7z_{3J@4al)gxgPL!=d~ zIPD%;W7daM+`6ivpf&ZrWyHokDQ9%hK^nL|LL@5E93kYfDV?^|FHTffg{5rDvPjFo zqX;)1{Osn#jSAZGW8eJSIS_lPwRsyp!zEy;v#}U9-^51x&2)vm&ruylwk4-Z$A5M! zP~f^J7WOu_*mvI1I$?UbLOk$ZwZE=MPC$SUX8;c>1vh!dFE;BX6Eog}#5mhrwq6Cy zY^rmjzZr8q*@p4s3)3~%S-Uqp;dduE8SL53JG@$FPaQlUoTWiao~U`5{y2hT7b49g zFEQ8N!mG0|C50R4eRlNE-jh|Uzz&SetlYm`2?A7aSG`h1%K*1Xc~zlNkSJyOxOoL7 za8~~LJW7=GyM#`unKD`EXb3c18XWGqjFR)V@%Q~rE_SGA#pyr0XK|3Z(DNm521X7Q zhCeV_*Zj73NsJP(&#u(}Y_sGg!_S~=FvUZ6NX(umPOdvowB^$t)+W;UI(<*UM@O4B z&Et%c!D>P5zUwkPD09dh$csU)j-}mei^rA9g{5$Ft4Q+uJuP5I9@j?TaCU+47)P;7 zLjc`yjBL4i)20a(H6j4dKa*%a_CqF6nny2|FR)01hJ-M5N+eT1Q^|_gKOodojGOvy=OhcN`ko6sNI+J9Tut`-R*|ga;K~X6;YXUGMa7%q zi2tAZ-8@iOPR$dzje!tAcv=loY1EQSwFrq$T0R5|`5mSZ8@A;~*h#89r-6Kd_g%xM z#iJJDWhhs;&3$*~6;wG6n@JG^QQO8WG#@Cgoz#t?) z!Bfatc~vgE$^CWwQ9zoE`TgmeBBPx|o5xNWoDtU)ik3@U3Fgu|d3_Si1wPQU!U9pX zhUs_s=g9V0@rgtgZM$fJ1lwO;%Q~Y|BHpI2r)kivrH1Rh2IXk;ns6RXn}aGr3yX}w z2m;sR{Ek2n2*;clGQ8#m8hqC~eSKZZly5fm@5M)E>unBR;!XNhq>ymBBo5HW&{|`R zjT`HGK01A6WtP?^&X#4ayNHM>XJ*&=2;*T`;E&bsDR|Z|#e-1gOLXFFAy=*qxL?Mi z@X<<4%3O4AQ;6T6L+S*n4xTCxe~ zY1XRg^aM(htM3Er&9tOy4)ZX%^VR?05+tSECX`@8D2B9`O42)~A`}qr{GMUdkwzShejZE^Oco%9J&0HkB1mGscvQdH4H5vgVin2EUETTp{dYhoUy6KNd-+m=7q z5T8}OU~^;uWiT3A6fLRD_Z^thc$+?YhpW4K(+u$JPIR@QX1BOll~-CTpG+sjJB9j1 zOL_e5MEb;?cQ+}-Q1Iq6>7YU=DSQz}K*xMZnNRjaWU%2c&!+#C3%JGVEO>Hr*UI-B z$3~AZQ4J>ZlZNeZp38Ud1+yN_G@b5$`S><{^V4ealXnq-!5zxfxPLgjzJ(#>RDwxP zgC6Ces0cfg(#6?U!{-oLU2>s6--F%s-^b9p^0ZMGOq1Gp34J?xu13gT<+D~xYwsp~sAQ~e%OoZyj@9JCI<=^qx>afE zQv}Qjcgq&d>s6PkMG*oR|J)C11C24U@hNF6nqMr}>=S_Vpi|isJzHWh3EiX!duBmC z6As?#`u0cCKDLu|7}e35jq9a0=Ay2);tl(`M#!?WN+Ol$v3i@#^b#EdNGHqyDXxV`*yrGcYIaH<6T?$Oy=7?BywZD#%(VvfI!Svosi@OpRFxNE}=xtpG47!45tw=6V zAJgcjQ+KGzg{URiGW+FYFQUx*SZgjkgApq#p&CY6x|q6IewJ_7wojGWA6xhGBk-mO zzf9S3&Tt1&O@h-VDs@O7X5&Y~7IW+HQMkUqkfd%P_2Yi7J-K`NQ$*uJ&IF1V&R0}n zxFi8DiC6S@a6Qp;)c7%Hsp7%WlA~q8Z}hL#+~hM)couofO47{jVzAP%TzJuRA=>X!{G%#rjDnJwrKa7|F7aDqtc6`l zJ9|tbrCbs|PrlALvqMPxAR$T2 zBPPzFHR^%%q#4+jDudoT&2+Cj3In#!pC}&>Q`cLS#j>6_%O_!R)LI4-7?I7zv`(*d z>hCB`c*AGKF^{y#(h)P(jhdv@Z@((rY*&khQ`L=4bFt+lqSN2bb=4TzA6M$o zkf~@#&6Um#zGg^Y66&CLwhR06)FcM#++NQxwo)S>XN?%)lXHc-?hNXM79lQZj%wONVnA?AR=Bc& z{I^Ph$V302diO7ipTO;#jVPH`KpDzl6{9YC!3$*3xg8ZWuuwZd1q$)3sI&2)`cPaG zQrb8~oJimRG}2I8c(|Un$T2KJ#R{INE?KufJCks^w9Sf#he7B+dU%vMZAg8B&PHF0Pv*N-M-17 zne%n|o3XM~+x`YgIt7l)o1(}yC5|qW!GE%xnny$zV?pep0h@^=M`c^H0!PGKT z-8RwJPW&6U#R^W&D>_W!soWnqt3|Tu-ZH_rWgtTBZb|DW6!+dy(xsb%k&+D932yd| zS`F`IQU?_Eg-IKyXvbQ{JTkl?zFu&Hp0-j~xPVe1uBqlV8;N9ywoTh4;>u6G*>0(> z5Gm(?4?j!QD!W&DtLD9{&24BM8;F$*EGQc6?kQYv-N(UWZt>WBS&?14+kIB?0QanEpfK>`LTR4cwXg25ysI z^O;)&5&}N0gq(AIxmm?$jCm&UlH>1*Wn&3l+6nDSEt5RB1C^?onOTRA|wU2f5Z-G-y}q;gfKTyN+3CzH&x# z^e~Uy-+Y(UfbujCPV-MLKO<32oXdqx>%53aoM!F0 zl0RiqkiZt+_kG@yzofTI6;>4@+q%Zw5ny>+q3~o(TnJcD)AtLr2eB~xhOk-mJ)ZS~ zy-i1XnJYe8!S$+hK|fzxpIx%7CF1n#^ZOrnOIbgm|8QsSTSQ@r5BCdbPNr+x;Cm1*w?@cUE1kzqKOAAPJN8odE8klxe?)bu_v50r&3K>cOjShtc z>Eh1C6a#`;Fl<}Io|%KSOC71092hR#96LAt@^f*2$%^A>W?{fWncoe#AP~g@SfOn! zU%GIAuDjT?PiGP-30FD>W5=#jX8ej94yJRCxegC)6sH|IzQuPoe!qLYLE_~Bg@L%7 zUKZtG+Y^3$5OnOdEs9@)*tcw#v^*I$NjkLgrQ|5tcqZZ$sYz&4k<5U2 z$P{Z2&ae%c&1?fp~oppcXR!Q>u`wmhrin6S~Q0R z+d1mFI2`Nre1O~JpvNS$`||kw^+=P45JPAaZ@FqRbL)Q*$!MPaeI&7XFw69=s@<1J z^KRGiV?I@RAf%>hST7~orDmr)Y}>lL6lQ`?thh4bHj2KWH&DjwpR#8u%?Ep%^PO(~ zmg>ne&-Sw^7a^fZIQBC8r7=wZz%&OU@u>eXis9s)GGHU@a=sz+88mY`hJAX;hDvguY#X_4m)JLSBpjpC($o*F~_`sWAYM8gQ0*iXL!_w9?Ok; zwk%sQoAuj`{qD8%9Vd3-{+k@vsgT^-iBVi~vBP=0VT{j=LfO(%?%2xMb!LZWD?;u%$)wul^n;vvxBa zh6TjEgWr8p6|7 后续修订:Sir 明确 serverless/scale-to-0 后,默认自建选择已由 [D8](../decisions.md) 替代;当前方向与缺失值判别见 [SaaS 修订报告](sentinel-saas-20261003.md)。本文件保留当时的实验事实和决策背景。 + +本轮补齐 G1 的两个判别:持久 Job carrier 是否能兼容实际旧消费者,以及分立后端能否形成一个可操作的诊断入口。结果支持进入实现;没有启动真实多 Peer 应用,没有使用生产数据,也没有证明生产容量。早期 OpenObserve 否决及 Tempo 重启/flags 限制仍以[前轮报告](README.md)为准。 + +## 数据库与传播容量 + +独立 PostgreSQL 使用仓库现有 `scripts/database.py init --profile development` 初始化,基线 readiness 通过。实验只在 disposable database 添加两个 nullable text 列,各用 `octet_length <= 512` 限制;未创建应用 migration。实际旧 Python JobRepository 创建记录时两列为 NULL,读取已有 carrier 后 claim/close 更新终态与 state,carrier 保持不变。实际旧 TypeScript JobManager、Job schema、DBAPIClient 经 PostgREST 完成同样操作;只替换配置/认证来源以指向实验库,没有替换 HTTP transport。这是 Node 运行证据,不是浏览器上下文证据。 + +语义损坏但有界的 `broken` 可以存入;513 bytes 被数据库约束拒绝。临时将 `public.alembic_version` 改成实验 head 后,实际旧 readiness 明确拒绝 migration mismatch,最后恢复原 head。因此新旧模型的数据兼容成立,任意旧新 runtime 混跑不成立。选择协调升级,不为本任务新增兼容版本框架。 + +Python SDK 注入的 traceparent 为 55 bytes;合法的 32 项 tracestate 可以达到 1109 bytes。512 是应用持久化容量政策,不是 W3C 最大值。生产 capture 应先让 SDK 标准化,再在存入 Job 前省略超限的整个 optional tracestate,保留 traceparent 的 ID 与 flags,记录有界省略原因;不自制 W3C parser,不截断半个成员,也不让观测状态导致业务事务失败。[W3C 的容量与截断规则](https://www.w3.org/TR/trace-context/) + +证据:[Python/数据库](evidence/convergence-20261003/database-probe.json)、[TypeScript/PostgREST](evidence/convergence-20261003/database-client-result.json)、[SDK 容量](evidence/convergence-20261003/carrier-capacity.json)。实验曾先误用 schema 限定的 alembic_version 路径;从真实 catalog 确认后修正任务脚本,仅重置独立实验数据再运行,没有变更业务仓库。 + +## 采集到查询 + +标准 Python SDK 发出 submit、execute 和 model spans、关联日志、counter=3、histogram count=2/sum=0.4;再发出 AI usage 缺失与真实零两条日志。Collector 用原生 OTLP/HTTP 出口分别写 Tempo、Loki、Prometheus。独立 API 和 Grafana datasource proxy 均读回预期结果。 + +| 判别 | 观察与适用边界 | +| --- | --- | +| 日志字段 | Job/trace/span 可关联;缺失 usage 不存在、零为字符串 `"0"`。Loki 把点换成下划线、非字符串属性字符串化;可消费但不承诺原始类型保真 | +| 指标 | cumulative counter=3;直方图 count=2、sum=0.4,各 bucket 正确。没有 Job ID 标签,不开启实验性 delta 转换 | +| Trace | Job 42 搜索到提交与执行两条 Trace;执行有模型子 span,Link 指向提交 span | +| Grafana 页面 | Job 42 显示三条日志,统计面板为 3 和 2;展开日志可打开执行 Trace,再点 `View linked span` 在分屏显示 job.submit | +| 角色 | anonymous Viewer 看得到固定面板但没有 Explore/内部 Trace 链接;仅此 loopback 合成实验改成 Editor 后链路通过。正式部署使用有身份的诊断用户,不能复制匿名配置 | + +内部日志链接从真实页面读取后在同一 tab 打开,因为实验浏览器的新窗口点击未创建 tab;后续 Span Link 实际点击成功。没有通过构造一个独立 API 查询冒充 UI 跳转。Grafana datasource provisioning 的 `${__value.raw}` 需要在 Grafana 与 Compose 两层分别转义;运行中 inline config 变更需重建对应容器才生效。 + +原始证据:[输入标识](evidence/convergence-20261003/expected.json)、[独立查询](evidence/convergence-20261003/stack-direct.json)、[Grafana proxy](evidence/convergence-20261003/stack-grafana.json)、[额外日志输入](evidence/convergence-20261003/stack-log-input.json)、[UI 观察](evidence/convergence-20261003/ui-observation.json)。主要依据:[Loki OTLP 映射](https://grafana.com/docs/loki/latest/send-data/otel/)、[Prometheus OTLP](https://prometheus.io/docs/guides/opentelemetry/)、[Loki derived fields](https://grafana.com/docs/grafana/latest/datasources/loki/configure/)、[Grafana Explore 角色配置](https://grafana.com/docs/grafana/latest/setup-grafana/configure-grafana/)。 + +## 版本与资源观察 + +本轮继续使用前轮 Collector 0.162.0、Tempo 3.1.0 和 Python OTel 1.45.0,新增镜像如下,完整可执行配置在 [stack-lab.py](stack-lab.py) 和 [database-lab.py](database-lab.py)。固定版本是实现起点,升级仍需重验字段映射。 + +| 组件 | 固定镜像 | +| --- | --- | +| Loki 3.7.8 | `grafana/loki@sha256:1107dd5274e0ada47e42472b7a7e71f3b2a2fe878878108f3e2f9e51528f0193` | +| Prometheus 3.15.0 | `prom/prometheus@sha256:efd719c99d83b060d9daefdcf00360461adf279f45ef5391f8d111892118753e` | +| Grafana 13.2.3 | `grafana/grafana@sha256:b28bae15e219c998fb0e0424ed724930cc61b1f61fb404d47c862f9a23f9e572` | +| pgvector pg17 | `pgvector/pgvector:pg17@sha256:d2ef61f42ef767baa5a1475393303cc235bcd92febd9d7014eddb48b41f3bad0` | +| PostgREST 14.15 | `postgrest/postgrest:v14.15@sha256:2f8e7b656f09db697a8875177694b417b35cb76c21370de07fc54e711e902326` | + +五组件限制合计 2 CPU/4 GiB:Tempo 和 Loki 各 0.5 CPU/1 GiB,Prometheus 0.35 CPU/768 MiB,Grafana 0.4 CPU/768 MiB,Collector 0.25 CPU/512 MiB。数据库实验另限 PostgreSQL 1 CPU/768 MiB、PostgREST 0.25 CPU/256 MiB,不计入观测栈预算。 + +UI 查询后的 [Docker 快照](evidence/convergence-20261003/stack-stats.jsonl)合计约 608 MiB,五个[容器状态](evidence/convergence-20261003/stack-state.jsonl)均未标记 OOM。十次极小日志查询的客户端 RTT 中位数约 9.71 ms,Grafana proxy 约 9.17 ms。它们包含本机经 SSH 到远端的路径,不能外推为生产 P95、CPU/RSS 峰值或每天磁盘增长。 + +Grafana 首次启动日志从 06:04:33 到 06:12:51 UTC 才开始 HTTP listen,约 8 分 18 秒,期间有一次后台资源注册超时;保留 SQLite 卷后的重建明显更快。Tempo idle scheduler 仍有 `no jobs found` 噪声。G3 必须复核冷启动预算、健康检查宽限和运维噪声。2 CPU/4 GiB 是可行性实验上限,不能写成最低配置或已满足 Sir 的实际预算。 + +## 复现与结束状态 + +从 core-py 根运行,先确认对应实验项目不存在冲突。`database-probe.py probe` 的候选 DDL 只供新初始化的实验库执行一次;重跑应新建 disposable 实验库,不指向任何共享库。 + +```bash +python3 tasks/observability-foundation/experiments/database-lab.py up +pdm run python tasks/observability-foundation/experiments/database-probe.py init +pdm run python tasks/observability-foundation/experiments/database-probe.py probe +python3 tasks/observability-foundation/experiments/database-lab.py postgrest +node tasks/observability-foundation/experiments/database-client.mjs +pdm run tasks/observability-foundation/experiments/carrier-capacity.py +python3 tasks/observability-foundation/experiments/stack-lab.py up +pdm run tasks/observability-foundation/experiments/signals.py +python3 tasks/observability-foundation/experiments/stack-probe.py send-log-cases +python3 tasks/observability-foundation/experiments/stack-probe.py direct +python3 tasks/observability-foundation/experiments/stack-probe.py grafana +python3 tasks/observability-foundation/experiments/database-lab.py stop +python3 tasks/observability-foundation/experiments/stack-lab.py stop +``` + +各服务 ready 与异步摄取可见后才运行依赖步骤;Grafana 首次初始化不能用一条短超时判为失败。连接方式与命令级 WSL interop 说明沿用前轮报告。新的 runtime 凭据与短时 JWT 均为 0600、处于忽略目录,不归档。 + +本轮结束停止两个项目的七个容器并关闭两个 SSH 隧道,保留独立合成数据卷:`inkcre-o11y-db-g1-b0a97f7c_data` 和 `inkcre-o11y-stack-g1-b0a97f7c_{tempo,loki,prometheus,grafana}`。加上前轮四个容器共十一项,最终状态见 [resource-final.json](evidence/convergence-20261003/resource-final.json)。不删除父任务数据,不操作 SVC 开发数据库。 + +G1 的设计判别已结束;正式入口认证、真实浏览器 context、应用故障隔离、内容出口 canary、实际 provider、容量/备份恢复继续作为 G2/G3 实现验收,不能借这份报告提前标通过。 diff --git a/tasks/observability-foundation/experiments/database-client.mjs b/tasks/observability-foundation/experiments/database-client.mjs new file mode 100644 index 0000000..f3f63a5 --- /dev/null +++ b/tasks/observability-foundation/experiments/database-client.mjs @@ -0,0 +1,44 @@ +// Exercise the actual old JobManager, Job schema and DBAPIClient against disposable PostgREST. +import assert from 'node:assert/strict'; +import { readFile, writeFile } from 'node:fs/promises'; +import { createRequire } from 'node:module'; +import { resolve, dirname } from 'node:path'; +import { fileURLToPath } from 'node:url'; + +const here = dirname(fileURLToPath(import.meta.url)); +const webRoot = resolve(here, '../../../../client-web'); +const { build } = createRequire(resolve(webRoot, 'package.json'))( + resolve(webRoot, 'node_modules/.pnpm/esbuild@0.28.1/node_modules/esbuild/lib/main.js') +); +globalThis.__o11yLab = JSON.parse(await readFile(resolve(here, 'runtime/database-client.json'), 'utf8')); +const bundle = await build({ + stdin: {contents: "export { Job } from './src/job/job'; export { JobManager } from './src/job/manager'; export { z } from 'zod';", resolveDir: resolve(webRoot, 'packages/core')}, + bundle: true, write: false, platform: 'node', format: 'esm', + plugins: [{name: 'local-runtime-config', setup(build) { + build.onResolve({filter: /^\.\.\/(config|auth)$/}, (args) => { + if (args.importer.endsWith('/base/db-api.ts')) return {path: args.path, namespace: 'lab'}; + }); + build.onLoad({filter: /.*/, namespace: 'lab'}, (args) => ({contents: args.path.endsWith('/config') + ? 'export const configStore = {metaConfig: {INKCRE_PGREST_URL: globalThis.__o11yLab.baseUrl}};' + : 'export const authStore = {getToken: async () => globalThis.__o11yLab.token};'})); + }}], +}); +const { Job, JobManager, z } = await import(`data:text/javascript;base64,${Buffer.from(bundle.outputFiles[0].text).toString('base64')}`); +JobManager.registerHandler('o11y-lab', { + parameters: z.object({}), canHandle: () => true, + handle: async (job) => {job.state = {synthetic: 'typescript-finished'};}, +}); +const old = await JobManager.create('o11y-lab', {}, 30); +const oldRaw = (await Job.dbApi.from().select().eq('id', old.id).single()).data; +assert.equal(oldRaw.submission_traceparent, null); +assert.equal(oldRaw.submission_tracestate, null); +const existing = await Job.get(globalThis.__o11yLab.jobId); +assert.equal(existing.submission_traceparent, undefined); +assert.equal(await JobManager.run(existing.id), true); +const raw = (await Job.dbApi.from().select().eq('id', existing.id).single()).data; +assert.equal(raw.status, 'finished'); +assert.deepEqual(raw.state, {synthetic: 'typescript-finished'}); +for (const [key, value] of Object.entries(globalThis.__o11yLab.carrier)) assert.equal(raw[key], value); +const report = {legacy_ts_insert_defaults_null: true, legacy_ts_read_ignores_carrier: true, legacy_ts_claim_close_preserve_carrier: true, transport: 'actual DBAPIClient and PostgREST', runtime: 'Node; config/auth sources replaced with dedicated lab values; browser not tested'}; +await writeFile(resolve(here, 'runtime/database-client-result.json'), JSON.stringify(report, null, 2) + '\n'); +console.log(JSON.stringify(report, null, 2)); diff --git a/tasks/observability-foundation/experiments/database-lab.py b/tasks/observability-foundation/experiments/database-lab.py new file mode 100644 index 0000000..508ed22 --- /dev/null +++ b/tasks/observability-foundation/experiments/database-lab.py @@ -0,0 +1,98 @@ +"""Disposable database compatibility lab; never addresses the SVC database project.""" + +import json +import os +from pathlib import Path +import secrets +import subprocess +import sys + +from lab import docker, HERE, HOST + +PROJECT = "inkcre-o11y-db-g1-b0a97f7c" +SOCKET = "/tmp/inkcre-o11y-db-b0a97f7c.sock" + + +def credentials(): + return json.loads((HERE / "runtime/database-credential.json").read_text()) + + +def compose(): + c = credentials() + return json.dumps( + { + "services": { + "postgres": { + "image": "pgvector/pgvector:pg17@sha256:d2ef61f42ef767baa5a1475393303cc235bcd92febd9d7014eddb48b41f3bad0", # noqa: E501 + "cpus": 1, + "mem_limit": "768m", + "environment": {"POSTGRES_DB": "o11y_lab", "POSTGRES_PASSWORD": c["admin"]}, + "ports": ["127.0.0.1:35432:5432"], + "volumes": ["data:/var/lib/postgresql/data"], + }, + "postgrest": { + "image": "postgrest/postgrest:v14.15@sha256:2f8e7b656f09db697a8875177694b417b35cb76c21370de07fc54e711e902326", # noqa: E501 + "cpus": 0.25, + "mem_limit": "256m", + "ports": ["127.0.0.1:33000:3000"], + "environment": { + "PGRST_DB_URI": f"postgresql://authenticator:{c['rest']}@postgres:5432/o11y_lab", + "PGRST_DB_SCHEMAS": "inkcre", + "PGRST_DB_ANON_ROLE": "anonymous", + "PGRST_DB_PRE_REQUEST": "inkcre_internal.check_jwt", + "PGRST_JWT_AUD": "inkcre-api", + "PGRST_JWT_SECRET": c["jwt"], + }, + }, + }, + "volumes": {"data": {}}, + } + ) + + +def main(action): + if action == "up": + path = HERE / "runtime/database-credential.json" + if not path.exists(): + with os.fdopen(os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600), "w") as f: + json.dump( + {key: secrets.token_urlsafe(32) for key in ["admin", "core", "rest", "jwt"]}, f + ) + print( + docker("compose", "-p", PROJECT, "-f", "-", "up", "-d", "postgres", data=compose()) + ) + if not Path(SOCKET).exists(): + subprocess.run( # noqa: S603 + [ + "ssh", + "-M", + "-S", + SOCKET, + "-fNT", + "-o", + "BatchMode=yes", + "-o", + "ExitOnForwardFailure=yes", + "-L", + "127.0.0.1:35432:127.0.0.1:35432", + "-L", + "127.0.0.1:33000:127.0.0.1:33000", + HOST, + ], + check=True, + ) + elif action == "postgrest": + print( + docker("compose", "-p", PROJECT, "-f", "-", "up", "-d", "postgrest", data=compose()) + ) + elif action in {"stop", "remove"}: + args = ["stop"] if action == "stop" else ["down", "--volumes"] + print(docker("compose", "-p", PROJECT, "-f", "-", *args, data=compose())) + if Path(SOCKET).exists(): + subprocess.run(["ssh", "-S", SOCKET, "-O", "exit", HOST], check=True) # noqa: S603 + else: + raise SystemExit("Use up, postgrest, stop, or remove") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tasks/observability-foundation/experiments/database-probe.py b/tasks/observability-foundation/experiments/database-probe.py new file mode 100644 index 0000000..9b7b99b --- /dev/null +++ b/tasks/observability-foundation/experiments/database-probe.py @@ -0,0 +1,173 @@ +"""Actual legacy native writes plus candidate DDL in the dedicated disposable database.""" + +import asyncio +import json +import os +from pathlib import Path +import subprocess +import sys + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +sys.path.insert(0, str(ROOT)) +CREDENTIALS = json.loads((HERE / "runtime/database-credential.json").read_text()) +ADMIN_URL = f"postgresql://postgres:{CREDENTIALS['admin']}@127.0.0.1:35432/o11y_lab" +CORE_URL = ( + f"postgresql+psycopg://inkcre_core:{CREDENTIALS['core']}@127.0.0.1:35432/o11y_lab" +) +# The dedicated database, port and credentials come only from this lab. Never use .env URLs. +os.environ.update( + MIGRATION_DATABASE_URL=ADMIN_URL, DATABASE_URL=CORE_URL, INKCRE_ENV_FILE="" +) + +import psycopg +from sqlalchemy.ext.asyncio import create_async_engine, AsyncSession + +from app.database_contract.readiness import check_database_contract +from app.persistence.job.repository import JobRepository +from app.schemas.job import JobModel, JobStatus +from scripts.verify_postgrest_contract import _token + + +async def native_writes(new_id): + engine = create_async_engine(CORE_URL) + try: + async with AsyncSession(engine) as session: + async with session.begin(): + repository = JobRepository(session) + old = await repository.create(JobModel(type="o11y-lab", timeout_seconds=30)) + old_id = old.id + new = await repository.get(new_id) + assert new is not None and new.status == JobStatus.PENDING + assert not hasattr(new, "submission_traceparent") + claimed = await repository.claim(new_id) + assert claimed is not None + claimed.state = {"synthetic": "native-finished"} + assert await repository.close(claimed, JobStatus.FINISHED) + return old_id + finally: + await engine.dispose() + + +def main(action): + if action == "init": + env = { + **os.environ, + "CORE_DATABASE_PASSWORD": CREDENTIALS["core"], + "POSTGREST_DATABASE_PASSWORD": CREDENTIALS["rest"], + } + result = subprocess.run( + [sys.executable, "scripts/database.py", "init", "--profile", "development"], + cwd=ROOT, + env=env, + check=False, + text=True, + capture_output=True, + ) + (HERE / "runtime/database-init.log").write_text(result.stdout + result.stderr) + assert result.returncode == 0, ( + "Initialization failed; inspect local ignored runtime log" + ) + print("Dedicated database initialized using existing lifecycle") + elif action == "probe": + baseline = check_database_contract("development", CORE_URL).as_dict() + assert baseline["status"] == "ok", baseline + carrier = { + "submission_traceparent": "00-" + "1" * 32 + "-" + "2" * 16 + "-01", + "submission_tracestate": "inkcre=synthetic", + } + with psycopg.connect(ADMIN_URL) as conn: + conn.execute( + "ALTER TABLE inkcre.jobs ADD COLUMN submission_traceparent text CHECK (octet_length(submission_traceparent) <= 512), ADD COLUMN submission_tracestate text CHECK (octet_length(submission_tracestate) <= 512)" # noqa: E501 + ) + conn.execute( + "INSERT INTO inkcre.job_types (id, description, parameters_schema, default_timeout_seconds) VALUES ('o11y-lab', 'synthetic compatibility probe', '{}', 30)" # noqa: E501 + ) + row = conn.execute( + "INSERT INTO inkcre.jobs (type, timeout_seconds, submission_traceparent, submission_tracestate) VALUES ('o11y-lab', 30, %s, %s) RETURNING id", # noqa: E501 + tuple(carrier.values()), + ).fetchone() + new_id = row[0] + old_id = asyncio.run(native_writes(new_id)) + with psycopg.connect(ADMIN_URL) as conn: + stored = conn.execute( + "SELECT status, state, submission_traceparent, submission_tracestate FROM inkcre.jobs WHERE id=%s", # noqa: E501 + (new_id,), + ).fetchone() + assert stored[0] == "finished" and stored[1] == {"synthetic": "native-finished"} + assert stored[2:] == tuple(carrier.values()) + assert conn.execute( + "SELECT submission_traceparent, submission_tracestate FROM inkcre.jobs WHERE id=%s", + (old_id,), + ).fetchone() == (None, None) + bounded = conn.execute( + "INSERT INTO inkcre.jobs (type, timeout_seconds, submission_traceparent) VALUES ('o11y-lab', 30, %s) RETURNING id", # noqa: E501 + ("broken",), + ).fetchone()[0] + try: + with conn.transaction(): + conn.execute( + "INSERT INTO inkcre.jobs (type, timeout_seconds, submission_tracestate) VALUES ('o11y-lab', 30, %s)", # noqa: E501 + ("x" * 513,), + ) + except psycopg.errors.CheckViolation: + capacity_rejected = True + else: + raise AssertionError("Oversized carrier not rejected") + original_heads = conn.execute( + "SELECT version_num FROM public.alembic_version" + ).fetchall() + assert len(original_heads) == 1 + conn.execute("UPDATE public.alembic_version SET version_num='o11y_probe_candidate'") + try: + next_schema = check_database_contract("development", CORE_URL).as_dict() + assert ( + next_schema["status"] == "error" and next_schema["migration"]["status"] == "error" + ) + finally: + with psycopg.connect(ADMIN_URL) as conn: + conn.execute("UPDATE public.alembic_version SET version_num=%s", original_heads[0]) + with psycopg.connect(ADMIN_URL) as conn: + ts_id = conn.execute( + "INSERT INTO inkcre.jobs (type, timeout_seconds, submission_traceparent, submission_tracestate) VALUES ('o11y-lab', 30, %s, %s) RETURNING id", # noqa: E501 + tuple(carrier.values()), + ).fetchone()[0] + conn.execute("NOTIFY pgrst, 'reload schema'") + # Short-lived synthetic credential used only by the local PostgREST probe, never + # archived. + payload = { + "baseUrl": "http://127.0.0.1:33000", + "token": _token(CREDENTIALS["jwt"]), + "jobId": ts_id, + "carrier": carrier, + } + path = HERE / "runtime/database-client.json" + with os.fdopen(os.open(path, os.O_CREAT | os.O_WRONLY | os.O_TRUNC, 0o600), "w") as f: + json.dump(payload, f) + report = { + "baseline": baseline, + "legacy_native_insert_defaults_null": True, + "legacy_native_claim_close_preserve_carrier": True, + "bounded_semantically_invalid_record": bounded, + "over_512_bytes_rejected": capacity_rejected, + "simulated_new_head_old_readiness": next_schema, + "head_restored": True, + "note": "Candidate DDL and synthetic head marker only; no production migration or full runtime boot", # noqa: E501 + } + (HERE / "runtime/database-probe.json").write_text(json.dumps(report, indent=2) + "\n") + print( + json.dumps( + { + key: value + for key, value in report.items() + if key not in {"baseline", "simulated_new_head_old_readiness"} + }, + indent=2, + ) + ) + else: + raise SystemExit("Use init or probe") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tasks/observability-foundation/experiments/evidence/application-probe.json b/tasks/observability-foundation/experiments/evidence/application-probe.json new file mode 100644 index 0000000..eda51db --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/application-probe.json @@ -0,0 +1,24 @@ +{ + "False": { + "ready": true, + "unauthenticated_rejected": true, + "relay_status": 503, + "shutdown_seconds": 0.616 + }, + "True": { + "ready": true, + "unauthenticated_rejected": true, + "relay_status": 200, + "shutdown_seconds": 0.543 + }, + "checks": [ + "default-off makes zero OTLP requests", + "real run.py readyz/lifespan both modes", + "existing Peer JWT required", + "protobuf preserved; authenticated JSON rejected with 415", + "private server authorization only upstream", + "400 malformed / 413 capacity / 502 upstream rejection", + "shared deployment identity present", + "no credentials in trace payload" + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/candidate-types.json b/tasks/observability-foundation/experiments/evidence/candidate-types.json new file mode 100644 index 0000000..6840299 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/candidate-types.json @@ -0,0 +1,7 @@ +{ + "source": "candidate migrated disposable o11y_impl, not production-admitted artifact", + "migration_head": "3d9593b0c855", + "generator": "supabase@2.112.0", + "sha256": "c80b2bccceb49f58f9708d262b3bd45a50f3d46ec75169cdf13d3aa1bc9ae398", + "required_before_merge": "regenerate/compare via sync-database-contract.mjs from admitted core artifact" +} diff --git a/tasks/observability-foundation/experiments/evidence/cloud-probe.json b/tasks/observability-foundation/experiments/evidence/cloud-probe.json new file mode 100644 index 0000000..1ca0bac --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/cloud-probe.json @@ -0,0 +1,22 @@ +{ + "run_id": "3e4b0adf-a931-4d29-8bda-778a4ce6c345", + "started_at": "2026-10-03T10:17:37.306333+00:00", + "submit_trace_id": "4374070441b7178095accb2645a5dc2e", + "execute_trace_id": "6fec6a9a9012ddc35cd01b69955c3630", + "http_responses": [ + { + "signal": "logs", + "status": 204 + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "metrics", + "status": 200 + } + ], + "all_signals_accepted": true, + "query_verified": false +} diff --git a/tasks/observability-foundation/experiments/evidence/cloud-query.json b/tasks/observability-foundation/experiments/evidence/cloud-query.json new file mode 100644 index 0000000..17e6349 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/cloud-query.json @@ -0,0 +1,158 @@ +{ + "run_id": "3e4b0adf-a931-4d29-8bda-778a4ce6c345", + "signals": { + "submit_trace_id": { + "status": 200 + }, + "execute_trace_id": { + "status": 200 + }, + "traces": { + "verified": true, + "span_count": 3 + }, + "logs": { + "status": 200, + "verified": true, + "events": [ + "job.closed", + "job.submitted" + ] + }, + "metrics": { + "status": 200, + "series": [ + { + "metric": { + "__name__": "inkcre_ai_token_usage_total", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "ai.chat", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance", + "token_type": "input" + }, + "value": [ + 1791022687, + "0" + ] + }, + { + "metric": { + "__name__": "inkcre_ai_usage_missing_total", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "ai.chat", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance", + "token_type": "output" + }, + "value": [ + 1791022687, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_operation_count_total", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "ai.chat", + "outcome": "success", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance" + }, + "value": [ + 1791022687, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_operation_count_total", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "job.execute", + "outcome": "success", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance" + }, + "value": [ + 1791022687, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_operation_count_total", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "job.submit", + "outcome": "success", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance" + }, + "value": [ + 1791022687, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_operation_duration_seconds_count", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "ai.chat", + "outcome": "success", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance" + }, + "value": [ + 1791022687, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_operation_duration_seconds_count", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "job.execute", + "outcome": "success", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance" + }, + "value": [ + 1791022687, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_operation_duration_seconds_count", + "instance": "b3de7599-1835-42d0-af12-c0d9baa77580", + "job": "core-py", + "operation": "job.submit", + "outcome": "success", + "service_instance_id": "b3de7599-1835-42d0-af12-c0d9baa77580", + "service_name": "core-py", + "service_version": "observability-acceptance" + }, + "value": [ + 1791022687, + "1" + ] + } + ], + "verified": true + } + }, + "query_verified": true +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/carrier-capacity.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/carrier-capacity.json new file mode 100644 index 0000000..f402d17 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/carrier-capacity.json @@ -0,0 +1,8 @@ +{ + "sdk_injected_traceparent_bytes": 55, + "sdk_injected_tracestate_bytes": 1109, + "candidate_max_bytes_each": 512, + "oversize_state_policy": "omit optional state as a whole; preserve parent IDs/flags", + "future_context_reinjected_as_version_00": true, + "persistence_limit_is_application_policy_not_W3C_maximum": true +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/database-client-result.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/database-client-result.json new file mode 100644 index 0000000..515af96 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/database-client-result.json @@ -0,0 +1,7 @@ +{ + "legacy_ts_insert_defaults_null": true, + "legacy_ts_read_ignores_carrier": true, + "legacy_ts_claim_close_preserve_carrier": true, + "transport": "actual DBAPIClient and PostgREST", + "runtime": "Node; config/auth sources replaced with dedicated lab values; browser not tested" +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/database-probe.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/database-probe.json new file mode 100644 index 0000000..05c166f --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/database-probe.json @@ -0,0 +1,82 @@ +{ + "baseline": { + "format": 1, + "status": "ok", + "profile": "development", + "contract": { + "revision": "peer-extension-setup-v1" + }, + "database": { + "status": "ok", + "environment": "development" + }, + "migration": { + "status": "ok", + "current": [ + "d41cc84db0c5" + ], + "expected": [ + "d41cc84db0c5" + ] + }, + "roles": { + "status": "ok", + "problems": [] + }, + "privileges": { + "status": "ok", + "problems": [] + }, + "catalog": { + "status": "ok", + "problems": [] + }, + "seed": { + "status": "ok", + "problems": [] + } + }, + "legacy_native_insert_defaults_null": true, + "legacy_native_claim_close_preserve_carrier": true, + "bounded_semantically_invalid_record": 7, + "over_512_bytes_rejected": true, + "simulated_new_head_old_readiness": { + "format": 1, + "status": "error", + "profile": "development", + "contract": { + "revision": "peer-extension-setup-v1" + }, + "database": { + "status": "ok", + "environment": "development" + }, + "migration": { + "status": "error", + "current": [ + "o11y_probe_candidate" + ], + "expected": [ + "d41cc84db0c5" + ] + }, + "roles": { + "status": "ok", + "problems": [] + }, + "privileges": { + "status": "ok", + "problems": [] + }, + "catalog": { + "status": "ok", + "problems": [] + }, + "seed": { + "status": "ok", + "problems": [] + } + }, + "head_restored": true, + "note": "Candidate DDL and synthetic head marker only; no production migration or full runtime boot" +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/expected.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/expected.json new file mode 100644 index 0000000..b586d57 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/expected.json @@ -0,0 +1,10 @@ +{ + "run_id": "b39448742f4c4d529765b5d8a5542c78", + "submission_trace_id": "55dec1d63651d535aff62dbe6d1c5868", + "submission_span_id": "c44ad3e5adab406f", + "execution_trace_id": "99c84282770a2129e78f7df2fede070a", + "execution_span_id": "ba9400e86e83fb4b", + "expected_job_counter": 3, + "expected_duration_count": 2, + "expected_duration_sum": 0.4 +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/packet-check.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/packet-check.json new file mode 100644 index 0000000..c19b5c7 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/packet-check.json @@ -0,0 +1,27 @@ +{ + "date": "2026-10-03", + "counts_checked": { + ".md": 12, + ".py": 14, + ".mjs": 3, + ".json": 21, + ".jsonl": 3 + }, + "syntax_and_json": "passed", + "local_markdown_links_fences_whitespace": "passed", + "hub_patch_apply_check": "passed", + "hub_patch_temporary_apply_and_links": "passed", + "hub_patch_files": [ + "20-product-tdd/observability-contract.md", + "docs/index.md", + "20-product-tdd/cross-unit-contracts.md", + "20-product-tdd/knowledge-capability-contract.md" + ], + "generated_svc_region": "unchanged", + "hub_and_shared_worktrees": "clean", + "core_tracked_files": "unchanged", + "evidence_invariants": "passed", + "isolated_resources": "11 stopped containers; 4 closed tunnels; volumes retained", + "full_repo_check": "not run: task packet and isolated experiments only", + "patch_sha256": "d4be67491b16f7c3096cf4d59a7b922f025245051a0461d3f16f3b348d1f0061" +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/resource-final.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/resource-final.json new file mode 100644 index 0000000..2377e50 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/resource-final.json @@ -0,0 +1,67 @@ +{ + "containers": [ + { + "Names": "inkcre-o11y-stack-g1-b0a97f7c-grafana-1", + "State": "exited", + "Status": "Exited (0) 7 minutes ago" + }, + { + "Names": "inkcre-o11y-db-g1-b0a97f7c-postgrest-1", + "State": "exited", + "Status": "Exited (255) 7 minutes ago" + }, + { + "Names": "inkcre-o11y-stack-g1-b0a97f7c-prometheus-1", + "State": "exited", + "Status": "Exited (0) 7 minutes ago" + }, + { + "Names": "inkcre-o11y-stack-g1-b0a97f7c-tempo-1", + "State": "exited", + "Status": "Exited (137) 7 minutes ago" + }, + { + "Names": "inkcre-o11y-stack-g1-b0a97f7c-loki-1", + "State": "exited", + "Status": "Exited (0) 7 minutes ago" + }, + { + "Names": "inkcre-o11y-stack-g1-b0a97f7c-collector-1", + "State": "exited", + "Status": "Exited (0) 7 minutes ago" + }, + { + "Names": "inkcre-o11y-db-g1-b0a97f7c-postgres-1", + "State": "exited", + "Status": "Exited (137) 7 minutes ago" + }, + { + "Names": "inkcre-o11y-tempo-g1-b0a97f7c-tempo-1", + "State": "exited", + "Status": "Exited (137) About an hour ago" + }, + { + "Names": "inkcre-o11y-tempo-g1-b0a97f7c-collector-1", + "State": "exited", + "Status": "Exited (0) About an hour ago" + }, + { + "Names": "inkcre-o11y-g1-b0a97f7c-openobserve-1", + "State": "exited", + "Status": "Exited (0) 40 hours ago" + }, + { + "Names": "inkcre-o11y-g1-b0a97f7c-collector-1", + "State": "exited", + "Status": "Exited (0) 40 hours ago" + } + ], + "sockets": { + "/tmp/inkcre-o11y-g1-b0a97f7c.sock": false, + "/tmp/inkcre-o11y-tempo-b0a97f7c.sock": false, + "/tmp/inkcre-o11y-db-b0a97f7c.sock": false, + "/tmp/inkcre-o11y-stack-b0a97f7c.sock": false + }, + "volumes": "retained; only synthetic lab data", + "svc_database": "not addressed" +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-direct.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-direct.json new file mode 100644 index 0000000..107ccf8 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-direct.json @@ -0,0 +1,789 @@ +{ + "expected": { + "run_id": "b39448742f4c4d529765b5d8a5542c78", + "submission_trace_id": "55dec1d63651d535aff62dbe6d1c5868", + "submission_span_id": "c44ad3e5adab406f", + "execution_trace_id": "99c84282770a2129e78f7df2fede070a", + "execution_span_id": "ba9400e86e83fb4b", + "expected_job_counter": 3, + "expected_duration_count": 2, + "expected_duration_sum": 0.4 + }, + "logs": { + "status": "success", + "data": { + "resultType": "streams", + "result": [ + { + "stream": { + "detected_level": "info", + "flags": "3", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_job_id": "42", + "inkcre_lab_run_id": "b39448742f4c4d529765b5d8a5542c78", + "inkcre_peer_id": "synthetic-python", + "scope_name": "inkcre-g1", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "severity_number": "9", + "span_id": "ba9400e86e83fb4b", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "trace_id": "99c84282770a2129e78f7df2fede070a" + }, + "values": [ + [ + "1791007710549609000", + "synthetic job finished" + ] + ] + }, + { + "stream": { + "detected_level": "unknown", + "gen_ai_usage_input_tokens": "0", + "inkcre_job_id": "42", + "inkcre_lab_case": "zero", + "service_name": "inkcre-o11y-synthetic", + "span_id": "ba9400e86e83fb4b", + "trace_id": "99c84282770a2129e78f7df2fede070a" + }, + "values": [ + [ + "1791007839670992001", + "synthetic zero" + ] + ] + }, + { + "stream": { + "detected_level": "unknown", + "inkcre_job_id": "42", + "inkcre_lab_case": "absent", + "service_name": "inkcre-o11y-synthetic", + "span_id": "ba9400e86e83fb4b", + "trace_id": "99c84282770a2129e78f7df2fede070a" + }, + "values": [ + [ + "1791007839670992000", + "synthetic absent" + ] + ] + } + ], + "stats": { + "summary": { + "bytesProcessedPerSecond": 88701, + "linesProcessedPerSecond": 1055, + "totalBytesProcessed": 252, + "totalLinesProcessed": 3, + "execTime": 0.002841, + "queueTime": 0.000141, + "subqueries": 0, + "totalEntriesReturned": 3, + "splits": 2, + "shards": 2, + "totalPostFilterLines": 3, + "totalStructuredMetadataBytesProcessed": 200 + }, + "querier": { + "store": { + "totalChunksRef": 0, + "totalChunksDownloaded": 0, + "chunksDownloadTime": 0, + "queryReferencedStructuredMetadata": false, + "queryUsedV2Engine": false, + "chunk": { + "headChunkBytes": 0, + "headChunkLines": 0, + "decompressedBytes": 0, + "decompressedLines": 0, + "compressedBytes": 0, + "totalDuplicates": 0, + "postFilterLines": 0, + "headChunkStructuredMetadataBytes": 0, + "decompressedStructuredMetadataBytes": 0 + }, + "chunkRefsFetchTime": 0, + "congestionControlLatency": 0, + "pipelineWrapperFilteredLines": 0, + "dataobj": { + "prePredicateDecompressedRows": 0, + "prePredicateDecompressedBytes": 0, + "prePredicateDecompressedStructuredMetadataBytes": 0, + "postPredicateRows": 0, + "postPredicateDecompressedBytes": 0, + "postPredicateStructuredMetadataBytes": 0, + "postFilterRows": 0, + "pagesScanned": 0, + "pagesDownloaded": 0, + "pagesDownloadedBytes": 0, + "pageBatches": 0, + "totalRowsAvailable": 0, + "totalPageDownloadTime": 0 + } + }, + "querierExecTime": 0.00377 + }, + "ingester": { + "totalReached": 2, + "totalChunksMatched": 2, + "totalBatches": 3, + "totalLinesSent": 3, + "store": { + "totalChunksRef": 0, + "totalChunksDownloaded": 0, + "chunksDownloadTime": 0, + "queryReferencedStructuredMetadata": true, + "queryUsedV2Engine": false, + "chunk": { + "headChunkBytes": 252, + "headChunkLines": 3, + "decompressedBytes": 0, + "decompressedLines": 0, + "compressedBytes": 0, + "totalDuplicates": 0, + "postFilterLines": 3, + "headChunkStructuredMetadataBytes": 200, + "decompressedStructuredMetadataBytes": 0 + }, + "chunkRefsFetchTime": 106203, + "congestionControlLatency": 0, + "pipelineWrapperFilteredLines": 0, + "dataobj": { + "prePredicateDecompressedRows": 0, + "prePredicateDecompressedBytes": 0, + "prePredicateDecompressedStructuredMetadataBytes": 0, + "postPredicateRows": 0, + "postPredicateDecompressedBytes": 0, + "postPredicateStructuredMetadataBytes": 0, + "postFilterRows": 0, + "pagesScanned": 0, + "pagesDownloaded": 0, + "pagesDownloadedBytes": 0, + "pageBatches": 0, + "totalRowsAvailable": 0, + "totalPageDownloadTime": 0 + } + }, + "recvWaitTime": 0.000965 + }, + "cache": { + "chunk": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "index": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "result": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "statsResult": { + "entriesFound": 0, + "entriesRequested": 1, + "entriesStored": 1, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 2, + "downloadTime": 19700, + "queryLengthServed": 0 + }, + "volumeResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "seriesResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "labelResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "instantMetricResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + } + }, + "index": { + "totalChunks": 0, + "postFilterChunks": 0, + "shardsDuration": 0, + "usedBloomFilters": false, + "totalStreams": 0, + "chunkRefsLookupTime": 0, + "bloomFilterTime": 0 + } + } + } + }, + "traces": { + "submission_trace_id": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "inkcre.deployment.id", + "value": { + "stringValue": "synthetic-g1" + } + }, + { + "key": "inkcre.peer.id", + "value": { + "stringValue": "synthetic-python" + } + }, + { + "key": "service.instance.id", + "value": { + "stringValue": "d999372d-0c5f-455d-8694-07f9882d33c5" + } + }, + { + "key": "service.version", + "value": { + "stringValue": "g1" + } + }, + { + "key": "telemetry.sdk.language", + "value": { + "stringValue": "python" + } + }, + { + "key": "telemetry.sdk.name", + "value": { + "stringValue": "opentelemetry" + } + }, + { + "key": "telemetry.sdk.version", + "value": { + "stringValue": "1.45.0" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-o11y-synthetic" + } + } + ] + }, + "scopeSpans": [ + { + "scope": { + "name": "inkcre-g1" + }, + "spans": [ + { + "traceId": "Vd7B1jZR1TWv9i2+bRxYaA==", + "spanId": "xErT5a2rQG8=", + "name": "job.submit", + "kind": "SPAN_KIND_INTERNAL", + "startTimeUnixNano": "1791007710523452000", + "endTimeUnixNano": "1791007710523468000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "b39448742f4c4d529765b5d8a5542c78" + } + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "47696" + } + }, + "execution_trace_id": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "inkcre.deployment.id", + "value": { + "stringValue": "synthetic-g1" + } + }, + { + "key": "inkcre.peer.id", + "value": { + "stringValue": "synthetic-python" + } + }, + { + "key": "service.instance.id", + "value": { + "stringValue": "d999372d-0c5f-455d-8694-07f9882d33c5" + } + }, + { + "key": "service.version", + "value": { + "stringValue": "g1" + } + }, + { + "key": "telemetry.sdk.language", + "value": { + "stringValue": "python" + } + }, + { + "key": "telemetry.sdk.name", + "value": { + "stringValue": "opentelemetry" + } + }, + { + "key": "telemetry.sdk.version", + "value": { + "stringValue": "1.45.0" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-o11y-synthetic" + } + } + ] + }, + "scopeSpans": [ + { + "scope": { + "name": "inkcre-g1" + }, + "spans": [ + { + "traceId": "mchCgncKISnnj33y/t4HCg==", + "spanId": "upQA6G6D+0s=", + "name": "job.execute", + "kind": "SPAN_KIND_INTERNAL", + "startTimeUnixNano": "1791007710541974000", + "endTimeUnixNano": "1791007710564276000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "b39448742f4c4d529765b5d8a5542c78" + } + } + ], + "links": [ + { + "traceId": "Vd7B1jZR1TWv9i2+bRxYaA==", + "spanId": "xErT5a2rQG8=" + } + ], + "status": {} + }, + { + "traceId": "mchCgncKISnnj33y/t4HCg==", + "spanId": "Rq3WKY8hf8s=", + "parentSpanId": "upQA6G6D+0s=", + "name": "chat synthetic-model", + "kind": "SPAN_KIND_INTERNAL", + "startTimeUnixNano": "1791007710542012000", + "endTimeUnixNano": "1791007710542016000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "b39448742f4c4d529765b5d8a5542c78" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.response.finish_reasons", + "value": { + "arrayValue": { + "values": [ + { + "stringValue": "stop" + } + ] + } + } + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "47696" + } + } + }, + "job_trace_search": { + "traces": [ + { + "traceID": "99c84282770a2129e78f7df2fede070a", + "rootServiceName": "inkcre-o11y-synthetic", + "rootTraceName": "job.execute", + "startTimeUnixNano": "1791007710541974000", + "durationMs": 22, + "spanSet": { + "spans": [ + { + "spanID": "46add6298f217fcb", + "startTimeUnixNano": "1791007710542012000", + "durationNanos": "4000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + }, + { + "spanID": "ba9400e86e83fb4b", + "startTimeUnixNano": "1791007710541974000", + "durationNanos": "22302000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 2 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "46add6298f217fcb", + "startTimeUnixNano": "1791007710542012000", + "durationNanos": "4000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + }, + { + "spanID": "ba9400e86e83fb4b", + "startTimeUnixNano": "1791007710541974000", + "durationNanos": "22302000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 2 + } + ], + "serviceStats": { + "inkcre-o11y-synthetic": { + "spanCount": 2 + } + } + }, + { + "traceID": "55dec1d63651d535aff62dbe6d1c5868", + "rootServiceName": "inkcre-o11y-synthetic", + "rootTraceName": "job.submit", + "startTimeUnixNano": "1791007710523452000", + "spanSet": { + "spans": [ + { + "spanID": "c44ad3e5adab406f", + "startTimeUnixNano": "1791007710523452000", + "durationNanos": "16000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "c44ad3e5adab406f", + "startTimeUnixNano": "1791007710523452000", + "durationNanos": "16000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-o11y-synthetic": { + "spanCount": 1 + } + } + } + ], + "metrics": { + "inspectedBytes": "15914", + "completedJobs": 3, + "totalJobs": 3 + } + }, + "metrics": { + "status": "success", + "data": { + "resultType": "vector", + "result": [ + { + "metric": { + "__name__": "inkcre_lab_jobs_total", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "outcome": "finished", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791007880.532, + "3" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_sum", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791007880.532, + "0.4" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_count", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791007880.532, + "2" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "0.1", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791007880.532, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "0.25", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791007880.532, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "0.5", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791007880.532, + "2" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "+Inf", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791007880.532, + "2" + ] + } + ] + } + }, + "log_query_roundtrip_ms": [ + 12.020791880786419, + 13.327708002179861, + 32.79262501746416, + 38.687500171363354, + 8.200999815016985, + 8.278375025838614, + 7.913082838058472, + 9.251666720956564, + 10.178124997764826, + 7.936624810099602 + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-grafana.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-grafana.json new file mode 100644 index 0000000..d3d9d15 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-grafana.json @@ -0,0 +1,789 @@ +{ + "expected": { + "run_id": "b39448742f4c4d529765b5d8a5542c78", + "submission_trace_id": "55dec1d63651d535aff62dbe6d1c5868", + "submission_span_id": "c44ad3e5adab406f", + "execution_trace_id": "99c84282770a2129e78f7df2fede070a", + "execution_span_id": "ba9400e86e83fb4b", + "expected_job_counter": 3, + "expected_duration_count": 2, + "expected_duration_sum": 0.4 + }, + "logs": { + "status": "success", + "data": { + "resultType": "streams", + "result": [ + { + "stream": { + "detected_level": "info", + "flags": "3", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_job_id": "42", + "inkcre_lab_run_id": "b39448742f4c4d529765b5d8a5542c78", + "inkcre_peer_id": "synthetic-python", + "scope_name": "inkcre-g1", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "severity_number": "9", + "span_id": "ba9400e86e83fb4b", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "trace_id": "99c84282770a2129e78f7df2fede070a" + }, + "values": [ + [ + "1791007710549609000", + "synthetic job finished" + ] + ] + }, + { + "stream": { + "detected_level": "unknown", + "gen_ai_usage_input_tokens": "0", + "inkcre_job_id": "42", + "inkcre_lab_case": "zero", + "service_name": "inkcre-o11y-synthetic", + "span_id": "ba9400e86e83fb4b", + "trace_id": "99c84282770a2129e78f7df2fede070a" + }, + "values": [ + [ + "1791007839670992001", + "synthetic zero" + ] + ] + }, + { + "stream": { + "detected_level": "unknown", + "inkcre_job_id": "42", + "inkcre_lab_case": "absent", + "service_name": "inkcre-o11y-synthetic", + "span_id": "ba9400e86e83fb4b", + "trace_id": "99c84282770a2129e78f7df2fede070a" + }, + "values": [ + [ + "1791007839670992000", + "synthetic absent" + ] + ] + } + ], + "stats": { + "summary": { + "bytesProcessedPerSecond": 95091, + "linesProcessedPerSecond": 1132, + "totalBytesProcessed": 252, + "totalLinesProcessed": 3, + "execTime": 0.00265, + "queueTime": 0.000242, + "subqueries": 0, + "totalEntriesReturned": 3, + "splits": 2, + "shards": 2, + "totalPostFilterLines": 3, + "totalStructuredMetadataBytesProcessed": 200 + }, + "querier": { + "store": { + "totalChunksRef": 0, + "totalChunksDownloaded": 0, + "chunksDownloadTime": 0, + "queryReferencedStructuredMetadata": false, + "queryUsedV2Engine": false, + "chunk": { + "headChunkBytes": 0, + "headChunkLines": 0, + "decompressedBytes": 0, + "decompressedLines": 0, + "compressedBytes": 0, + "totalDuplicates": 0, + "postFilterLines": 0, + "headChunkStructuredMetadataBytes": 0, + "decompressedStructuredMetadataBytes": 0 + }, + "chunkRefsFetchTime": 0, + "congestionControlLatency": 0, + "pipelineWrapperFilteredLines": 0, + "dataobj": { + "prePredicateDecompressedRows": 0, + "prePredicateDecompressedBytes": 0, + "prePredicateDecompressedStructuredMetadataBytes": 0, + "postPredicateRows": 0, + "postPredicateDecompressedBytes": 0, + "postPredicateStructuredMetadataBytes": 0, + "postFilterRows": 0, + "pagesScanned": 0, + "pagesDownloaded": 0, + "pagesDownloadedBytes": 0, + "pageBatches": 0, + "totalRowsAvailable": 0, + "totalPageDownloadTime": 0 + } + }, + "querierExecTime": 0.003721 + }, + "ingester": { + "totalReached": 2, + "totalChunksMatched": 2, + "totalBatches": 3, + "totalLinesSent": 3, + "store": { + "totalChunksRef": 0, + "totalChunksDownloaded": 0, + "chunksDownloadTime": 0, + "queryReferencedStructuredMetadata": true, + "queryUsedV2Engine": false, + "chunk": { + "headChunkBytes": 252, + "headChunkLines": 3, + "decompressedBytes": 0, + "decompressedLines": 0, + "compressedBytes": 0, + "totalDuplicates": 0, + "postFilterLines": 3, + "headChunkStructuredMetadataBytes": 200, + "decompressedStructuredMetadataBytes": 0 + }, + "chunkRefsFetchTime": 116412, + "congestionControlLatency": 0, + "pipelineWrapperFilteredLines": 0, + "dataobj": { + "prePredicateDecompressedRows": 0, + "prePredicateDecompressedBytes": 0, + "prePredicateDecompressedStructuredMetadataBytes": 0, + "postPredicateRows": 0, + "postPredicateDecompressedBytes": 0, + "postPredicateStructuredMetadataBytes": 0, + "postFilterRows": 0, + "pagesScanned": 0, + "pagesDownloaded": 0, + "pagesDownloadedBytes": 0, + "pageBatches": 0, + "totalRowsAvailable": 0, + "totalPageDownloadTime": 0 + } + }, + "recvWaitTime": 0.000716 + }, + "cache": { + "chunk": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "index": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "result": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "statsResult": { + "entriesFound": 1, + "entriesRequested": 1, + "entriesStored": 0, + "bytesReceived": 233, + "bytesSent": 0, + "requests": 1, + "downloadTime": 9701, + "queryLengthServed": 2732000000000 + }, + "volumeResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "seriesResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "labelResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + }, + "instantMetricResult": { + "entriesFound": 0, + "entriesRequested": 0, + "entriesStored": 0, + "bytesReceived": 0, + "bytesSent": 0, + "requests": 0, + "downloadTime": 0, + "queryLengthServed": 0 + } + }, + "index": { + "totalChunks": 0, + "postFilterChunks": 0, + "shardsDuration": 0, + "usedBloomFilters": false, + "totalStreams": 0, + "chunkRefsLookupTime": 0, + "bloomFilterTime": 0 + } + } + } + }, + "traces": { + "submission_trace_id": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "inkcre.deployment.id", + "value": { + "stringValue": "synthetic-g1" + } + }, + { + "key": "inkcre.peer.id", + "value": { + "stringValue": "synthetic-python" + } + }, + { + "key": "service.instance.id", + "value": { + "stringValue": "d999372d-0c5f-455d-8694-07f9882d33c5" + } + }, + { + "key": "service.version", + "value": { + "stringValue": "g1" + } + }, + { + "key": "telemetry.sdk.language", + "value": { + "stringValue": "python" + } + }, + { + "key": "telemetry.sdk.name", + "value": { + "stringValue": "opentelemetry" + } + }, + { + "key": "telemetry.sdk.version", + "value": { + "stringValue": "1.45.0" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-o11y-synthetic" + } + } + ] + }, + "scopeSpans": [ + { + "scope": { + "name": "inkcre-g1" + }, + "spans": [ + { + "traceId": "Vd7B1jZR1TWv9i2+bRxYaA==", + "spanId": "xErT5a2rQG8=", + "name": "job.submit", + "kind": "SPAN_KIND_INTERNAL", + "startTimeUnixNano": "1791007710523452000", + "endTimeUnixNano": "1791007710523468000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "b39448742f4c4d529765b5d8a5542c78" + } + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "47696" + } + }, + "execution_trace_id": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "inkcre.deployment.id", + "value": { + "stringValue": "synthetic-g1" + } + }, + { + "key": "inkcre.peer.id", + "value": { + "stringValue": "synthetic-python" + } + }, + { + "key": "service.instance.id", + "value": { + "stringValue": "d999372d-0c5f-455d-8694-07f9882d33c5" + } + }, + { + "key": "service.version", + "value": { + "stringValue": "g1" + } + }, + { + "key": "telemetry.sdk.language", + "value": { + "stringValue": "python" + } + }, + { + "key": "telemetry.sdk.name", + "value": { + "stringValue": "opentelemetry" + } + }, + { + "key": "telemetry.sdk.version", + "value": { + "stringValue": "1.45.0" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-o11y-synthetic" + } + } + ] + }, + "scopeSpans": [ + { + "scope": { + "name": "inkcre-g1" + }, + "spans": [ + { + "traceId": "mchCgncKISnnj33y/t4HCg==", + "spanId": "upQA6G6D+0s=", + "name": "job.execute", + "kind": "SPAN_KIND_INTERNAL", + "startTimeUnixNano": "1791007710541974000", + "endTimeUnixNano": "1791007710564276000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "b39448742f4c4d529765b5d8a5542c78" + } + } + ], + "links": [ + { + "traceId": "Vd7B1jZR1TWv9i2+bRxYaA==", + "spanId": "xErT5a2rQG8=" + } + ], + "status": {} + }, + { + "traceId": "mchCgncKISnnj33y/t4HCg==", + "spanId": "Rq3WKY8hf8s=", + "parentSpanId": "upQA6G6D+0s=", + "name": "chat synthetic-model", + "kind": "SPAN_KIND_INTERNAL", + "startTimeUnixNano": "1791007710542012000", + "endTimeUnixNano": "1791007710542016000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "b39448742f4c4d529765b5d8a5542c78" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.response.finish_reasons", + "value": { + "arrayValue": { + "values": [ + { + "stringValue": "stop" + } + ] + } + } + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "47696" + } + } + }, + "job_trace_search": { + "traces": [ + { + "traceID": "99c84282770a2129e78f7df2fede070a", + "rootServiceName": "inkcre-o11y-synthetic", + "rootTraceName": "job.execute", + "startTimeUnixNano": "1791007710541974000", + "durationMs": 22, + "spanSet": { + "spans": [ + { + "spanID": "46add6298f217fcb", + "startTimeUnixNano": "1791007710542012000", + "durationNanos": "4000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + }, + { + "spanID": "ba9400e86e83fb4b", + "startTimeUnixNano": "1791007710541974000", + "durationNanos": "22302000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 2 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "46add6298f217fcb", + "startTimeUnixNano": "1791007710542012000", + "durationNanos": "4000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + }, + { + "spanID": "ba9400e86e83fb4b", + "startTimeUnixNano": "1791007710541974000", + "durationNanos": "22302000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 2 + } + ], + "serviceStats": { + "inkcre-o11y-synthetic": { + "spanCount": 2 + } + } + }, + { + "traceID": "55dec1d63651d535aff62dbe6d1c5868", + "rootServiceName": "inkcre-o11y-synthetic", + "rootTraceName": "job.submit", + "startTimeUnixNano": "1791007710523452000", + "spanSet": { + "spans": [ + { + "spanID": "c44ad3e5adab406f", + "startTimeUnixNano": "1791007710523452000", + "durationNanos": "16000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "c44ad3e5adab406f", + "startTimeUnixNano": "1791007710523452000", + "durationNanos": "16000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-o11y-synthetic": { + "spanCount": 1 + } + } + } + ], + "metrics": { + "inspectedBytes": "15914", + "completedJobs": 3, + "totalJobs": 3 + } + }, + "metrics": { + "status": "success", + "data": { + "resultType": "vector", + "result": [ + { + "metric": { + "__name__": "inkcre_lab_jobs_total", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "outcome": "finished", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791008099.775, + "3" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_sum", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791008099.775, + "0.4" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_count", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791008099.775, + "2" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "0.1", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791008099.775, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "0.25", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791008099.775, + "1" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "0.5", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791008099.775, + "2" + ] + }, + { + "metric": { + "__name__": "inkcre_lab_duration_seconds_bucket", + "inkcre_deployment_id": "synthetic-g1", + "instance": "d999372d-0c5f-455d-8694-07f9882d33c5", + "job": "inkcre-o11y-synthetic", + "le": "+Inf", + "operation": "synthetic", + "service_instance_id": "d999372d-0c5f-455d-8694-07f9882d33c5", + "service_name": "inkcre-o11y-synthetic" + }, + "value": [ + 1791008099.775, + "2" + ] + } + ] + } + }, + "log_query_roundtrip_ms": [ + 50.10420875623822, + 8.892999961972237, + 8.73695919290185, + 9.863167069852352, + 11.442167218774557, + 8.665250148624182, + 8.47870809957385, + 8.920082822442055, + 9.424667339771986, + 10.566083248704672 + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-log-input.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-log-input.json new file mode 100644 index 0000000..0fb1b34 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-log-input.json @@ -0,0 +1,72 @@ +{ + "resourceLogs": [ + { + "resource": { + "attributes": [ + { + "key": "service.name", + "value": { + "stringValue": "inkcre-o11y-synthetic" + } + } + ] + }, + "scopeLogs": [ + { + "logRecords": [ + { + "timeUnixNano": "1791007839670992000", + "body": { + "stringValue": "synthetic absent" + }, + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.case", + "value": { + "stringValue": "absent" + } + } + ], + "traceId": "99c84282770a2129e78f7df2fede070a", + "spanId": "ba9400e86e83fb4b" + }, + { + "timeUnixNano": "1791007839670992001", + "body": { + "stringValue": "synthetic zero" + }, + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + }, + { + "key": "inkcre.lab.case", + "value": { + "stringValue": "zero" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "0" + } + } + ], + "traceId": "99c84282770a2129e78f7df2fede070a", + "spanId": "ba9400e86e83fb4b" + } + ] + } + ] + } + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-state.jsonl b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-state.jsonl new file mode 100644 index 0000000..ee86a0f --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-state.jsonl @@ -0,0 +1,5 @@ +{"Status":"running","Running":true,"Paused":false,"Restarting":false,"OOMKilled":false,"Dead":false,"Pid":39402,"ExitCode":0,"Error":"","StartedAt":"2026-10-03T06:04:33.484105705Z","FinishedAt":"0001-01-01T00:00:00Z"} +{"Status":"running","Running":true,"Paused":false,"Restarting":false,"OOMKilled":false,"Dead":false,"Pid":39255,"ExitCode":0,"Error":"","StartedAt":"2026-10-03T06:04:32.460827425Z","FinishedAt":"0001-01-01T00:00:00Z"} +{"Status":"running","Running":true,"Paused":false,"Restarting":false,"OOMKilled":false,"Dead":false,"Pid":39364,"ExitCode":0,"Error":"","StartedAt":"2026-10-03T06:04:33.339677681Z","FinishedAt":"0001-01-01T00:00:00Z"} +{"Status":"running","Running":true,"Paused":false,"Restarting":false,"OOMKilled":false,"Dead":false,"Pid":55767,"ExitCode":0,"Error":"","StartedAt":"2026-10-03T06:24:57.155839441Z","FinishedAt":"0001-01-01T00:00:00Z"} +{"Status":"running","Running":true,"Paused":false,"Restarting":false,"OOMKilled":false,"Dead":false,"Pid":39216,"ExitCode":0,"Error":"","StartedAt":"2026-10-03T06:04:31.881415766Z","FinishedAt":"0001-01-01T00:00:00Z"} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-stats.jsonl b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-stats.jsonl new file mode 100644 index 0000000..f7b1b6e --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/stack-stats.jsonl @@ -0,0 +1,5 @@ +{"BlockIO":"38.4MB / 446kB","CPUPerc":"0.24%","Container":"inkcre-o11y-stack-g1-b0a97f7c-tempo-1","ID":"a5373acfbe07","MemPerc":"5.75%","MemUsage":"58.84MiB / 1GiB","Name":"inkcre-o11y-stack-g1-b0a97f7c-tempo-1","NetIO":"16.6kB / 20.1kB","PIDs":"8"} +{"BlockIO":"9.47MB / 229kB","CPUPerc":"0.63%","Container":"inkcre-o11y-stack-g1-b0a97f7c-loki-1","ID":"1d3cb54d05db","MemPerc":"6.52%","MemUsage":"66.73MiB / 1GiB","Name":"inkcre-o11y-stack-g1-b0a97f7c-loki-1","NetIO":"33kB / 141kB","PIDs":"8"} +{"BlockIO":"101MB / 311kB","CPUPerc":"0.14%","Container":"inkcre-o11y-stack-g1-b0a97f7c-prometheus-1","ID":"f884b29a24b7","MemPerc":"14.43%","MemUsage":"110.8MiB / 768MiB","Name":"inkcre-o11y-stack-g1-b0a97f7c-prometheus-1","NetIO":"174kB / 60.5kB","PIDs":"7"} +{"BlockIO":"38.3MB / 725kB","CPUPerc":"0.84%","Container":"inkcre-o11y-stack-g1-b0a97f7c-grafana-1","ID":"d0a491323746","MemPerc":"40.50%","MemUsage":"311MiB / 768MiB","Name":"inkcre-o11y-stack-g1-b0a97f7c-grafana-1","NetIO":"95.5kB / 2.06MB","PIDs":"102"} +{"BlockIO":"78.2MB / 0B","CPUPerc":"0.04%","Container":"inkcre-o11y-stack-g1-b0a97f7c-collector-1","ID":"d7faaadbc21c","MemPerc":"11.80%","MemUsage":"60.39MiB / 512MiB","Name":"inkcre-o11y-stack-g1-b0a97f7c-collector-1","NetIO":"60.1kB / 164kB","PIDs":"7"} diff --git a/tasks/observability-foundation/experiments/evidence/convergence-20261003/ui-observation.json b/tasks/observability-foundation/experiments/evidence/convergence-20261003/ui-observation.json new file mode 100644 index 0000000..f4da78d --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/convergence-20261003/ui-observation.json @@ -0,0 +1,31 @@ +{ + "date": "2026-10-03", + "source": "CUA browser accessibility observations; synthetic loopback lab", + "dashboard": { + "job_id": 42, + "log_rows": 3, + "counter": 3, + "histogram_count": 2 + }, + "viewer": { + "explore_visible": false, + "derived_trace_link_visible": false + }, + "editor": { + "explore_visible": true, + "derived_trace_link_visible": true + }, + "execution": { + "trace_id": "99c84282770a2129e78f7df2fede070a", + "span_count": 2, + "root": "job.execute", + "child": "chat synthetic-model" + }, + "linked_submission": { + "trace_id": "55dec1d63651d535aff62dbe6d1c5868", + "span_id": "c44ad3e5adab406f", + "root": "job.submit", + "span_count": 1 + }, + "navigation_note": "Derived link href opened in same tab after new-window click did not create a tab in IAB; View linked span then opened submission in split pane." +} diff --git a/tasks/observability-foundation/experiments/evidence/foundation-probe.json b/tasks/observability-foundation/experiments/evidence/foundation-probe.json new file mode 100644 index 0000000..50f98ca --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/foundation-probe.json @@ -0,0 +1,105 @@ +{ + "off_no_threads_or_exports": true, + "three_signals": true, + "canary_absent": true, + "usage_known_zero_missing": true, + "exception_and_cancellation_propagate": true, + "restart": true, + "invalid_configuration_isolated": true, + "healthy_shutdown_seconds": 0.004106916952878237, + "slow_receiver_backlog_shutdown_seconds": 1.974071207921952, + "refused_receiver_shutdown_seconds": 0.9573320411145687, + "backlog_business_seconds": 0.07602291693910956, + "requests": 26, + "native_pipeline_metric_names": [ + "otel.sdk.exporter.log.exported", + "otel.sdk.exporter.log.inflight", + "otel.sdk.exporter.operation.duration", + "otel.sdk.exporter.span.exported", + "otel.sdk.exporter.span.inflight", + "otel.sdk.log.created", + "otel.sdk.processor.log.processed", + "otel.sdk.processor.log.queue.capacity", + "otel.sdk.processor.log.queue.size", + "otel.sdk.processor.span.processed", + "otel.sdk.processor.span.queue.capacity", + "otel.sdk.processor.span.queue.size", + "otel.sdk.span.live", + "otel.sdk.span.started" + ], + "native_queue_drops": { + "otel.sdk.processor.log.processed": 512, + "otel.sdk.processor.span.processed": 512 + }, + "native_export_failures": { + "otel.sdk.exporter.log.exported": 512, + "otel.sdk.exporter.span.exported": 512 + }, + "native_metric_attributes_and_exemplars_safe": true, + "fault_injection_checks": [ + "traces_initialization_rollback", + "logs_initialization_rollback", + "metrics_initialization_rollback", + "failed_signal_not_published_with_other_signals_active", + "start_span_failure_isolated", + "context_activation_failure_isolated", + "log_and_usage_failure_isolated", + "finalization_preserves_result_exception_and_cancellation" + ], + "timeout_configuration": { + "default": { + "traces": 10.0, + "logs": 10.0, + "metrics": 10.0 + }, + "default_slow_receiver": { + "delay_seconds": 1.2, + "successful_spans": 1 + }, + "dotenv_signal_and_environment_common": { + "traces": 3.5, + "logs": 4.0, + "metrics": 6.0 + }, + "environment_signal_over_dotenv": { + "traces": 7.0, + "logs": 4.0, + "metrics": 6.0 + }, + "invalid_0": { + "traces": null, + "logs": 4.0, + "metrics": 6.0 + }, + "invalid_-1": { + "traces": null, + "logs": 4.0, + "metrics": 6.0 + }, + "invalid_31": { + "traces": null, + "logs": 4.0, + "metrics": 6.0 + }, + "invalid_nan": { + "traces": null, + "logs": 4.0, + "metrics": 6.0 + }, + "invalid_inf": { + "traces": null, + "logs": 4.0, + "metrics": 6.0 + }, + "invalid_invalid": { + "traces": null, + "logs": 4.0, + "metrics": 6.0 + }, + "valid_signal_over_invalid_common": { + "traces": 30.0, + "logs": null, + "metrics": 6.0 + } + } +} diff --git a/tasks/observability-foundation/experiments/evidence/migration-roundtrip.json b/tasks/observability-foundation/experiments/evidence/migration-roundtrip.json new file mode 100644 index 0000000..76abc98 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/migration-roundtrip.json @@ -0,0 +1,28 @@ +{ + "accepted_512_byte_carriers": true, + "carrier_columns_and_capacity_checks_present": true, + "database": "o11y_migration_roundtrip", + "downgrade_head": "a0465e3b028f", + "downgrade_removes_carriers_as_expected": true, + "initial_head": "3d9593b0c855", + "legacy_job_and_pg_log_preserved": { + "after_downgrade": true, + "after_first_upgrade": true, + "after_reupgrade": true + }, + "legacy_job_id": 1, + "legacy_pg_log_id": 1, + "rejected_513_byte_carriers": { + "after_first_upgrade": [ + "submission_traceparent", + "submission_tracestate" + ], + "after_reupgrade": [ + "submission_traceparent", + "submission_tracestate" + ] + }, + "reupgrade_head": "3d9593b0c855", + "reupgrade_restores_columns_with_null_legacy_values": true, + "status": "passed" +} diff --git a/tasks/observability-foundation/experiments/evidence/openobserve-20261001/ai-projection.json b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/ai-projection.json new file mode 100644 index 0000000..eaf1b24 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/ai-projection.json @@ -0,0 +1,419 @@ +{ + "input": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + }, + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + } + ] + }, + "scopeSpans": [ + { + "spans": [ + { + "name": "submission", + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "startTimeUnixNano": "1790863448762958000", + "endTimeUnixNano": "1790863448762959000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + } + ] + }, + { + "name": "absent", + "traceId": "74046adf072c4fb98f00ff8a45206f9e", + "spanId": "8a31626387484695", + "startTimeUnixNano": "1790863448762958000", + "endTimeUnixNano": "1790863448762959000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + } + ], + "links": [ + { + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "traceState": "inkcre=synthetic", + "flags": 1, + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ] + }, + { + "name": "explicit", + "traceId": "8f5d0e572ac7400f8b9a74a64221564d", + "spanId": "f0abc9e591da4dfa", + "startTimeUnixNano": "1790863448762958000", + "endTimeUnixNano": "1790863448762959000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.agent.version", + "value": { + "stringValue": "agent-7" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.125 + } + } + ], + "links": [ + { + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "traceState": "inkcre=synthetic", + "flags": 1, + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ] + }, + { + "name": "zero", + "traceId": "870d82356c67456b971a98d82395edde", + "spanId": "f8f495dd55794f75", + "startTimeUnixNano": "1790863448762958000", + "endTimeUnixNano": "1790863448762959000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.0 + } + } + ], + "links": [ + { + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "traceState": "inkcre=synthetic", + "flags": 1, + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ] + } + ] + } + ] + } + ] + }, + "output": { + "took": 14, + "took_detail": { + "total": 14, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 13 + }, + "hits": [ + { + "_o2_ingest_ts": 1790863448644892, + "_timestamp": 1790863448762958, + "duration": 1, + "end_time": 1790863448762959000, + "events": "[]", + "flags": 1, + "gen_ai_agent_version": "service-3", + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_usage_cost": -0.0, + "gen_ai_usage_input_tokens": 0, + "gen_ai_usage_output_tokens": 0, + "gen_ai_usage_total_tokens": 0, + "inkcre_lab_run_id": "20d50f4624f645d8bf2ee385473de7f0", + "links": "[{\"context\":{\"traceId\":\"99cf4f054710442097ec1b830809ed22\",\"spanId\":\"b4842e47da3649aa\",\"traceFlags\":1,\"traceState\":\"inkcre=synthetic\"},\"inkcre.link.reason\":\"job-submission\",\"inkcre.link.sequence\":\"7\",\"droppedAttributesCount\":0}]", + "operation_name": "absent", + "service_name": "inkcre-projection-probe", + "service_service_version": "service-3", + "span_id": "8a31626387484695", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1790863448762958000, + "status_code": 0, + "status_message": "", + "trace_id": "74046adf072c4fb98f00ff8a45206f9e" + }, + { + "_o2_ingest_ts": 1790863448644976, + "_timestamp": 1790863448762958, + "duration": 1, + "end_time": 1790863448762959000, + "events": "[]", + "flags": 1, + "gen_ai_agent_version": "agent-7", + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_usage_cost": 0.125, + "gen_ai_usage_input_tokens": 8, + "gen_ai_usage_output_tokens": 3, + "gen_ai_usage_total_tokens": 11, + "inkcre_lab_run_id": "20d50f4624f645d8bf2ee385473de7f0", + "links": "[{\"context\":{\"traceId\":\"99cf4f054710442097ec1b830809ed22\",\"spanId\":\"b4842e47da3649aa\",\"traceFlags\":1,\"traceState\":\"inkcre=synthetic\"},\"inkcre.link.reason\":\"job-submission\",\"inkcre.link.sequence\":\"7\",\"droppedAttributesCount\":0}]", + "operation_name": "explicit", + "service_name": "inkcre-projection-probe", + "service_service_version": "service-3", + "span_id": "f0abc9e591da4dfa", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1790863448762958000, + "status_code": 0, + "status_message": "", + "trace_id": "8f5d0e572ac7400f8b9a74a64221564d" + }, + { + "_o2_ingest_ts": 1790863448644997, + "_timestamp": 1790863448762958, + "duration": 1, + "end_time": 1790863448762959000, + "events": "[]", + "flags": 1, + "gen_ai_agent_version": "service-3", + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_usage_cost": 0.0, + "gen_ai_usage_input_tokens": 0, + "gen_ai_usage_output_tokens": 0, + "gen_ai_usage_total_tokens": 0, + "inkcre_lab_run_id": "20d50f4624f645d8bf2ee385473de7f0", + "links": "[{\"context\":{\"traceId\":\"99cf4f054710442097ec1b830809ed22\",\"spanId\":\"b4842e47da3649aa\",\"traceFlags\":1,\"traceState\":\"inkcre=synthetic\"},\"inkcre.link.sequence\":\"7\",\"inkcre.link.reason\":\"job-submission\",\"droppedAttributesCount\":0}]", + "operation_name": "zero", + "service_name": "inkcre-projection-probe", + "service_service_version": "service-3", + "span_id": "f8f495dd55794f75", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1790863448762958000, + "status_code": 0, + "status_message": "", + "trace_id": "870d82356c67456b971a98d82395edde" + }, + { + "_o2_ingest_ts": 1790863448644781, + "_timestamp": 1790863448762958, + "duration": 1, + "end_time": 1790863448762959000, + "events": "[]", + "flags": 1, + "inkcre_lab_run_id": "20d50f4624f645d8bf2ee385473de7f0", + "links": "[]", + "operation_name": "submission", + "service_name": "inkcre-projection-probe", + "service_service_version": "service-3", + "span_id": "b4842e47da3649aa", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1790863448762958000, + "status_code": 0, + "status_message": "", + "trace_id": "99cf4f054710442097ec1b830809ed22" + } + ], + "total": 4, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 15, + "trace_id": "01a0f7c7a25d73d281bd3c9cc2232097", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": true, + "peak_memory_usage": 5.0 + }, + "linked_submission": { + "took": 217, + "took_detail": { + "total": 217, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 217 + }, + "hits": [ + { + "_o2_ingest_ts": 1790863448644781, + "_timestamp": 1790863448762958, + "duration": 1, + "end_time": 1790863448762959000, + "events": "[]", + "flags": 1, + "inkcre_lab_run_id": "20d50f4624f645d8bf2ee385473de7f0", + "links": "[]", + "operation_name": "submission", + "service_name": "inkcre-projection-probe", + "service_service_version": "service-3", + "span_id": "b4842e47da3649aa", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1790863448762958000, + "status_code": 0, + "status_message": "", + "trace_id": "99cf4f054710442097ec1b830809ed22" + } + ], + "total": 1, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 15, + "trace_id": "01a0f7c7a27378b380e2367ba2a8fcff", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": true, + "peak_memory_usage": 5.0 + } +} diff --git a/tasks/observability-foundation/experiments/evidence/openobserve-20261001/expected.json b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/expected.json new file mode 100644 index 0000000..fe48af1 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/expected.json @@ -0,0 +1,10 @@ +{ + "run_id": "61198a2de0444192bab7740dcf1191cc", + "submission_trace_id": "0e95c2d4ef6a1aae66a3a3a81f9e7132", + "submission_span_id": "cf2e8421e3769776", + "execution_trace_id": "d971b5bfdde8c0ce63be0a66c02991ba", + "execution_span_id": "fdb8090585876861", + "expected_job_counter": 3, + "expected_duration_count": 2, + "expected_duration_sum": 0.4 +} diff --git a/tasks/observability-foundation/experiments/evidence/openobserve-20261001/failure.json b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/failure.json new file mode 100644 index 0000000..c3bfead --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/failure.json @@ -0,0 +1,105 @@ +{ + "submitted_spans": 4096, + "before": "# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 0\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 0\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.39\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 7.7819904e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 9.603488e+06\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 2.1940128e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 2.7105544e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 415.679219289\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 4\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 12\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 4\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 7\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 7\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 7\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 7\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "samples": [ + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.5\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.424896e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 1.7966184e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 5.7889384e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 425.573184364\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.51\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.447424e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 1.9414232e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 5.9337432e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 426.581621407\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.51\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.447424e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 1.9619272e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 5.9542472e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 427.591738759\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.51\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4445568e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 1.9858104e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 5.9781304e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 428.598801013\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.52\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4457856e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.0476984e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.0400184e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 429.614473423\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.52\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4457856e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.0719784e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.0642984e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 430.626160797\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.52\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4457856e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.0958472e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.0881672e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 431.634781485\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.52\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4490624e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.119788e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.112108e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 432.640188241\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 912150\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.53\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4507008e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.2627704e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.2550904e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 433.648629016\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 608100\n# HELP otelcol_exporter_send_failed_spans Number of spans in failed attempts to send to destination. At detailed telemetry level, includes attributes: error.type (semantic convention), error.permanent. [Alpha]\n# TYPE otelcol_exporter_send_failed_spans counter\notelcol_exporter_send_failed_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 128\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.53\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.457664e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.3252584e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.3175784e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 434.658667008\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 608100\n# HELP otelcol_exporter_send_failed_spans Number of spans in failed attempts to send to destination. At detailed telemetry level, includes attributes: error.type (semantic convention), error.permanent. [Alpha]\n# TYPE otelcol_exporter_send_failed_spans counter\notelcol_exporter_send_failed_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 128\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.54\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4687232e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.3500632e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.3423832e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5494152e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 435.667172288\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 608100\n# HELP otelcol_exporter_send_failed_spans Number of spans in failed attempts to send to destination. At detailed telemetry level, includes attributes: error.type (semantic convention), error.permanent. [Alpha]\n# TYPE otelcol_exporter_send_failed_spans counter\notelcol_exporter_send_failed_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 128\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.54\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4691328e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.4121032e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.4044232e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5756296e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 436.675083525\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 608100\n# HELP otelcol_exporter_send_failed_spans Number of spans in failed attempts to send to destination. At detailed telemetry level, includes attributes: error.type (semantic convention), error.permanent. [Alpha]\n# TYPE otelcol_exporter_send_failed_spans counter\notelcol_exporter_send_failed_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 128\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.55\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4695424e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.4368704e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.4291904e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5756296e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 437.680841806\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 608100\n# HELP otelcol_exporter_send_failed_spans Number of spans in failed attempts to send to destination. At detailed telemetry level, includes attributes: error.type (semantic convention), error.permanent. [Alpha]\n# TYPE otelcol_exporter_send_failed_spans counter\notelcol_exporter_send_failed_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 128\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.55\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.4695424e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.4612584e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.4535784e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.5756296e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 438.688999077\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 1\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 608100\n# HELP otelcol_exporter_send_failed_spans Number of spans in failed attempts to send to destination. At detailed telemetry level, includes attributes: error.type (semantic convention), error.permanent. [Alpha]\n# TYPE otelcol_exporter_send_failed_spans counter\notelcol_exporter_send_failed_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 128\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.55\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.5031296e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 2.5233328e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.5156528e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.99506e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 439.697323677\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n" + ], + "after": "# HELP otelcol_exporter_enqueue_failed_spans Number of spans failed to be added to the sending queue. [Alpha]\n# TYPE otelcol_exporter_enqueue_failed_spans counter\notelcol_exporter_enqueue_failed_spans{exporter=\"otlp_http\"} 3712\n# HELP otelcol_exporter_in_flight_requests Number of export requests currently in-flight (including retry backoff). [Development]\n# TYPE otelcol_exporter_in_flight_requests gauge\notelcol_exporter_in_flight_requests{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_in_flight_requests{data_type=\"traces\",exporter=\"otlp_http\"} 0\n# HELP otelcol_exporter_queue_capacity Fixed capacity of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_capacity gauge\notelcol_exporter_queue_capacity{data_type=\"logs\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"metrics\",exporter=\"otlp_http\"} 1.048576e+06\notelcol_exporter_queue_capacity{data_type=\"traces\",exporter=\"otlp_http\"} 1.048576e+06\n# HELP otelcol_exporter_queue_size Current size of the retry queue (in batches). [Alpha]\n# TYPE otelcol_exporter_queue_size gauge\notelcol_exporter_queue_size{data_type=\"logs\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"metrics\",exporter=\"otlp_http\"} 0\notelcol_exporter_queue_size{data_type=\"traces\",exporter=\"otlp_http\"} 0\n# HELP otelcol_exporter_send_failed_spans Number of spans in failed attempts to send to destination. At detailed telemetry level, includes attributes: error.type (semantic convention), error.permanent. [Alpha]\n# TYPE otelcol_exporter_send_failed_spans counter\notelcol_exporter_send_failed_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 384\n# HELP otelcol_exporter_sent_log_records Number of log record successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_log_records counter\notelcol_exporter_sent_log_records{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/logs\"} 1\n# HELP otelcol_exporter_sent_metric_points Number of metric points successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_metric_points counter\notelcol_exporter_sent_metric_points{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/metrics\"} 4\n# HELP otelcol_exporter_sent_spans Number of spans successfully sent to destination. [Alpha]\n# TYPE otelcol_exporter_sent_spans counter\notelcol_exporter_sent_spans{exporter=\"otlp_http\",server_address=\"openobserve\",server_port=\"5080\",url_path=\"/api/default/v1/traces\"} 7\n# HELP otelcol_process_cpu_seconds Total CPU user and system time in seconds [Alpha]\n# TYPE otelcol_process_cpu_seconds counter\notelcol_process_cpu_seconds 0.5800000000000001\n# HELP otelcol_process_memory_rss Total physical memory (resident set size) [Alpha]\n# TYPE otelcol_process_memory_rss gauge\notelcol_process_memory_rss 9.5596544e+07\n# HELP otelcol_process_runtime_heap_alloc_bytes Bytes of allocated heap objects (see 'go doc runtime.MemStats.HeapAlloc') [Alpha]\n# TYPE otelcol_process_runtime_heap_alloc_bytes gauge\notelcol_process_runtime_heap_alloc_bytes 1.3929904e+07\n# HELP otelcol_process_runtime_total_alloc_bytes Cumulative bytes allocated for heap objects (see 'go doc runtime.MemStats.TotalAlloc') [Alpha]\n# TYPE otelcol_process_runtime_total_alloc_bytes counter\notelcol_process_runtime_total_alloc_bytes 6.7769304e+07\n# HELP otelcol_process_runtime_total_sys_memory_bytes Total bytes of memory obtained from the OS (see 'go doc runtime.MemStats.Sys') [Alpha]\n# TYPE otelcol_process_runtime_total_sys_memory_bytes gauge\notelcol_process_runtime_total_sys_memory_bytes 3.99506e+07\n# HELP otelcol_process_uptime Uptime of the process [Alpha]\n# TYPE otelcol_process_uptime counter\notelcol_process_uptime 451.849005114\n# HELP otelcol_processor_batch_batch_send_size Number of units in the batch [Development]\n# TYPE otelcol_processor_batch_batch_send_size histogram\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"25\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"75\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100\"} 4\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"250\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"500\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"750\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"1000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"2000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"3000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"4000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"5000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"6000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"7000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"8000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"9000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"10000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"20000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"30000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"50000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"100000\"} 7\notelcol_processor_batch_batch_send_size_bucket{processor=\"batch\",le=\"+Inf\"} 7\notelcol_processor_batch_batch_send_size_sum{processor=\"batch\"} 396\notelcol_processor_batch_batch_send_size_count{processor=\"batch\"} 7\n# HELP otelcol_processor_batch_batch_size_trigger_send Number of times the batch was sent due to a size trigger [Development]\n# TYPE otelcol_processor_batch_batch_size_trigger_send counter\notelcol_processor_batch_batch_size_trigger_send{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_metadata_cardinality Number of distinct metadata value combinations being processed [Development]\n# TYPE otelcol_processor_batch_metadata_cardinality gauge\notelcol_processor_batch_metadata_cardinality{processor=\"batch\"} 3\n# HELP otelcol_processor_batch_timeout_trigger_send Number of times the batch was sent due to a timeout trigger [Development]\n# TYPE otelcol_processor_batch_timeout_trigger_send counter\notelcol_processor_batch_timeout_trigger_send{processor=\"batch\"} 4\n# HELP otelcol_processor_incoming_items Number of items passed to the processor. [Alpha]\n# TYPE otelcol_processor_incoming_items counter\notelcol_processor_incoming_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_incoming_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_incoming_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_memory_limiter_accepted_log_records Number of log records successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_log_records counter\notelcol_processor_memory_limiter_accepted_log_records{processor=\"memory_limiter\"} 1\n# HELP otelcol_processor_memory_limiter_accepted_metric_points Number of metric points successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_metric_points counter\notelcol_processor_memory_limiter_accepted_metric_points{processor=\"memory_limiter\"} 4\n# HELP otelcol_processor_memory_limiter_accepted_spans Number of spans successfully pushed into the next component in the pipeline. [Alpha]\n# TYPE otelcol_processor_memory_limiter_accepted_spans counter\notelcol_processor_memory_limiter_accepted_spans{processor=\"memory_limiter\"} 4103\n# HELP otelcol_processor_outgoing_items Number of items emitted from the processor. [Alpha]\n# TYPE otelcol_processor_outgoing_items counter\notelcol_processor_outgoing_items{otel_signal=\"logs\",processor=\"memory_limiter\"} 1\notelcol_processor_outgoing_items{otel_signal=\"metrics\",processor=\"memory_limiter\"} 4\notelcol_processor_outgoing_items{otel_signal=\"traces\",processor=\"memory_limiter\"} 4103\n# HELP otelcol_receiver_accepted_log_records Number of log records successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_log_records counter\notelcol_receiver_accepted_log_records{receiver=\"otlp\",transport=\"http\"} 1\n# HELP otelcol_receiver_accepted_metric_points Number of metric points successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_metric_points counter\notelcol_receiver_accepted_metric_points{receiver=\"otlp\",transport=\"http\"} 4\n# HELP otelcol_receiver_accepted_spans Number of spans successfully pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_accepted_spans counter\notelcol_receiver_accepted_spans{receiver=\"otlp\",transport=\"http\"} 4103\n# HELP otelcol_receiver_failed_log_records The number of log records that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_log_records counter\notelcol_receiver_failed_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_metric_points The number of metric points that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_metric_points counter\notelcol_receiver_failed_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_failed_spans The number of spans that failed to be processed by the receiver due to internal errors. [Alpha]\n# TYPE otelcol_receiver_failed_spans counter\notelcol_receiver_failed_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_log_records Number of log records that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_log_records counter\notelcol_receiver_refused_log_records{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_metric_points Number of metric points that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_metric_points counter\notelcol_receiver_refused_metric_points{receiver=\"otlp\",transport=\"http\"} 0\n# HELP otelcol_receiver_refused_spans Number of spans that could not be pushed into the pipeline. [Alpha]\n# TYPE otelcol_receiver_refused_spans counter\notelcol_receiver_refused_spans{receiver=\"otlp\",transport=\"http\"} 0\n# HELP promhttp_metric_handler_errors_total Total number of internal errors encountered by the promhttp metric handler.\n# TYPE promhttp_metric_handler_errors_total counter\npromhttp_metric_handler_errors_total{cause=\"encoding\"} 0\npromhttp_metric_handler_errors_total{cause=\"gathering\"} 0\n# HELP target_info Target metadata\n# TYPE target_info gauge\ntarget_info{service_instance_id=\"a40c1db2-9495-4eef-a710-086e7b9b0f97\",service_name=\"otelcol\",service_version=\"0.162.0\"} 1\n", + "responses": [ + { + "request": 0, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 1, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 2, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 3, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 4, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 5, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 6, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 7, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 8, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 9, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 10, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 11, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 12, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 13, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 14, + "status": 200, + "body": "{\"partialSuccess\":{}}" + }, + { + "request": 15, + "status": 200, + "body": "{\"partialSuccess\":{}}" + } + ], + "collector_stats_after_burst": "{\"BlockIO\":\"112MB / 0B\",\"CPUPerc\":\"0.81%\",\"Container\":\"inkcre-o11y-g1-b0a97f7c-collector-1\",\"ID\":\"8475a22efb0f\",\"MemPerc\":\"20.37%\",\"MemUsage\":\"104.3MiB / 512MiB\",\"Name\":\"inkcre-o11y-g1-b0a97f7c-collector-1\",\"NetIO\":\"10.5MB / 234kB\",\"PIDs\":\"8\"}\n" +} diff --git a/tasks/observability-foundation/experiments/evidence/openobserve-20261001/propagation.json b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/propagation.json new file mode 100644 index 0000000..d0c24d7 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/propagation.json @@ -0,0 +1,17 @@ +{ + "vectors": [ + "sampled", + "unsampled", + "future-version", + "missing", + "malformed", + "zero-id", + "bad-state" + ], + "python_to_js_to_python": "passed", + "independent_execution_links": "passed", + "python_async_context_isolation": "passed", + "js_explicit_context_isolation": "passed", + "sdk_invalid_state_diagnostic_contains_raw_value": true, + "browser_runtime": "not tested" +} diff --git a/tasks/observability-foundation/experiments/evidence/openobserve-20261001/readback.json b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/readback.json new file mode 100644 index 0000000..ed6b355 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/openobserve-20261001/readback.json @@ -0,0 +1,568 @@ +{ + "expected": { + "run_id": "61198a2de0444192bab7740dcf1191cc", + "submission_trace_id": "0e95c2d4ef6a1aae66a3a3a81f9e7132", + "submission_span_id": "cf2e8421e3769776", + "execution_trace_id": "d971b5bfdde8c0ce63be0a66c02991ba", + "execution_span_id": "fdb8090585876861", + "expected_job_counter": 3, + "expected_duration_count": 2, + "expected_duration_sum": 0.4 + }, + "traces": { + "took": 12, + "took_detail": { + "total": 12, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 12 + }, + "hits": [ + { + "_o2_ingest_ts": 1790862762976932, + "_timestamp": 1790862763664021, + "duration": 11, + "end_time": 1790862763664032000, + "events": "[]", + "flags": 1, + "gen_ai_agent_version": "g1", + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_response_finish_reasons": "[\"stop\"]", + "gen_ai_usage_cost": -0.0, + "gen_ai_usage_input_tokens": 8, + "gen_ai_usage_output_tokens": 3, + "gen_ai_usage_total_tokens": 11, + "inkcre_job_id": "42", + "inkcre_lab_run_id": "61198a2de0444192bab7740dcf1191cc", + "links": "[]", + "operation_name": "chat synthetic-model", + "reference_parent_span_id": "fdb8090585876861", + "reference_parent_trace_id": "d971b5bfdde8c0ce63be0a66c02991ba", + "reference_ref_type": "ChildOf", + "service_inkcre_deployment_id": "synthetic-g1", + "service_inkcre_peer_id": "synthetic-python", + "service_name": "inkcre-o11y-synthetic", + "service_service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_service_version": "g1", + "service_telemetry_sdk_language": "python", + "service_telemetry_sdk_name": "opentelemetry", + "service_telemetry_sdk_version": "1.45.0", + "span_id": "4bb6456d6eac71df", + "span_kind": "1", + "span_status": "UNSET", + "start_time": 1790862763664021000, + "status_code": 0, + "status_message": "", + "trace_id": "d971b5bfdde8c0ce63be0a66c02991ba" + }, + { + "_o2_ingest_ts": 1790862762976955, + "_timestamp": 1790862763663933, + "duration": 9592, + "end_time": 1790862763673525000, + "events": "[]", + "flags": 1, + "inkcre_job_id": "42", + "inkcre_lab_run_id": "61198a2de0444192bab7740dcf1191cc", + "links": "[{\"context\":{\"traceId\":\"0e95c2d4ef6a1aae66a3a3a81f9e7132\",\"spanId\":\"cf2e8421e3769776\",\"traceFlags\":256,\"traceState\":\"\"},\"droppedAttributesCount\":0}]", + "operation_name": "job.execute", + "service_inkcre_deployment_id": "synthetic-g1", + "service_inkcre_peer_id": "synthetic-python", + "service_name": "inkcre-o11y-synthetic", + "service_service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_service_version": "g1", + "service_telemetry_sdk_language": "python", + "service_telemetry_sdk_name": "opentelemetry", + "service_telemetry_sdk_version": "1.45.0", + "span_id": "fdb8090585876861", + "span_kind": "1", + "span_status": "UNSET", + "start_time": 1790862763663933000, + "status_code": 0, + "status_message": "", + "trace_id": "d971b5bfdde8c0ce63be0a66c02991ba" + }, + { + "_o2_ingest_ts": 1790862762976831, + "_timestamp": 1790862763053168, + "duration": 22, + "end_time": 1790862763053190000, + "events": "[]", + "flags": 1, + "inkcre_job_id": "42", + "inkcre_lab_run_id": "61198a2de0444192bab7740dcf1191cc", + "links": "[]", + "operation_name": "job.submit", + "service_inkcre_deployment_id": "synthetic-g1", + "service_inkcre_peer_id": "synthetic-python", + "service_name": "inkcre-o11y-synthetic", + "service_service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_service_version": "g1", + "service_telemetry_sdk_language": "python", + "service_telemetry_sdk_name": "opentelemetry", + "service_telemetry_sdk_version": "1.45.0", + "span_id": "cf2e8421e3769776", + "span_kind": "1", + "span_status": "UNSET", + "start_time": 1790862763053168000, + "status_code": 0, + "status_message": "", + "trace_id": "0e95c2d4ef6a1aae66a3a3a81f9e7132" + } + ], + "total": 3, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 15, + "trace_id": "01a0f7c7a3d37442a15f75a0785882b2", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": true, + "peak_memory_usage": 0.0 + }, + "logs": { + "took": 86, + "took_detail": { + "total": 86, + "cache_took": 0, + "file_list_took": 61, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 24 + }, + "hits": [ + { + "_timestamp": 1790862763667939, + "body": "synthetic job finished", + "dropped_attributes_count": 0, + "inkcre_deployment_id": "synthetic-g1", + "inkcre_job_id": 42, + "inkcre_lab_run_id": "61198a2de0444192bab7740dcf1191cc", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "severity": 9, + "span_id": "fdb8090585876861", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "trace_id": "d971b5bfdde8c0ce63be0a66c02991ba" + } + ], + "total": 1, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 1, + "trace_id": "01a0f7c7a45b71428927ef1f0c330b51", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": true, + "peak_memory_usage": 0.0 + }, + "metrics": { + "inkcre_lab_jobs": { + "took": 8, + "took_detail": { + "total": 8, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 7 + }, + "hits": [ + { + "__hash__": 8020257630917457365, + "__name__": "inkcre_lab_jobs", + "_timestamp": 1790862763685748, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "is_monotonic": "true", + "outcome": "finished", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763677984000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 3.0 + } + ], + "total": 1, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 2, + "trace_id": "01a0f7c7a4e3753384f6686ae673511f", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": false, + "peak_memory_usage": 0.0 + }, + "inkcre_lab_duration_count": { + "took": 8, + "took_detail": { + "total": 8, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 7 + }, + "hits": [ + { + "__hash__": 4143641218075448382, + "__name__": "inkcre_lab_duration_count", + "_timestamp": 1790862763685748, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 2.0 + } + ], + "total": 1, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 2, + "trace_id": "01a0f7c7a51b7850a9c22b56f6d91e32", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": false, + "peak_memory_usage": 0.0 + }, + "inkcre_lab_duration_sum": { + "took": 7, + "took_detail": { + "total": 7, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 6 + }, + "hits": [ + { + "__hash__": 1990536794487974906, + "__name__": "inkcre_lab_duration_sum", + "_timestamp": 1790862763685748, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 0.4 + } + ], + "total": 1, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 2, + "trace_id": "01a0f7c7a52a7132a1b5af98dcbfcf28", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": false, + "peak_memory_usage": 0.0 + } + }, + "buckets": { + "took": 8, + "took_detail": { + "total": 8, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 7 + }, + "hits": [ + { + "__hash__": 12276402949810835769, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763685748, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "0.5", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 2.0 + }, + { + "__hash__": 850570520348517362, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763685748, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "inf", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 2.0 + }, + { + "__hash__": 7695652644745743727, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763685748, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "0.1", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 1.0 + }, + { + "__hash__": 1419446133234569247, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763685748, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "0.25", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 1.0 + }, + { + "__hash__": 850570520348517362, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763678238, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "inf", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 2.0 + }, + { + "__hash__": 12276402949810835769, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763678238, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "0.5", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 2.0 + }, + { + "__hash__": 1419446133234569247, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763678238, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "0.25", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 1.0 + }, + { + "__hash__": 7695652644745743727, + "__name__": "inkcre_lab_duration_bucket", + "_timestamp": 1790862763678238, + "aggregation_temporality": "AGGREGATION_TEMPORALITY_CUMULATIVE", + "flag": "DATA_POINT_FLAGS_DO_NOT_USE", + "inkcre_deployment_id": "synthetic-g1", + "inkcre_peer_id": "synthetic-python", + "instrumentation_library_name": "inkcre-g1", + "instrumentation_library_version": "", + "le": "0.1", + "operation": "synthetic", + "service_instance_id": "995fb155-ebf3-4e5a-b669-393ee423d02b", + "service_name": "inkcre-o11y-synthetic", + "service_version": "g1", + "start_time": "1790862763678183000", + "telemetry_sdk_language": "python", + "telemetry_sdk_name": "opentelemetry", + "telemetry_sdk_version": "1.45.0", + "value": 1.0 + } + ], + "total": 8, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 8, + "trace_id": "01a0f7c7a5637e80a5ab81d14897b0b8", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": true, + "peak_memory_usage": 0.0 + }, + "query_roundtrip_ms": [ + 60.588416177779436, + 61.10099982470274, + 59.29458420723677, + 14.071834273636341, + 15.679250005632639, + 63.86862508952618, + 57.845584116876125, + 55.933792144060135, + 56.127042043954134, + 58.49091615527868, + 56.892958004027605, + 13.800499960780144, + 13.834583573043346, + 13.305957894772291, + 13.618375174701214, + 15.724916011095047, + 63.00679221749306, + 58.84483316913247, + 57.32791591435671, + 15.848417300730944 + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/runtime-probe-metrics-only.json b/tasks/observability-foundation/experiments/evidence/runtime-probe-metrics-only.json new file mode 100644 index 0000000..838fe9e --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/runtime-probe-metrics-only.json @@ -0,0 +1,68 @@ +{ + "run_id": "e69c9f6d45094daa9eec1b57b4fb78bb", + "signal_mode": "metrics-only", + "operation_counts": { + "job.submit:success": 4, + "job.execute:success": 2, + "job.execute:error": 1, + "job.execute:cancelled": 1, + "agent.tool:error": 1, + "agent.tool:success": 1, + "http.server:success": 1, + "http.server:error": 2, + "peer.http:success": 1, + "peer.http:error": 2 + }, + "duration_sample_counts": { + "job.submit:success": 4, + "job.execute:success": 2, + "job.execute:error": 1, + "job.execute:cancelled": 1, + "agent.tool:error": 1, + "agent.tool:success": 1, + "http.server:success": 1, + "http.server:error": 2, + "peer.http:success": 1, + "peer.http:error": 2 + }, + "concurrent_tool_outcomes_isolated": true, + "scope": "disposable PostgreSQL + production Job/PG log/Peer HTTP/ASGI mechanisms; not full runtime or SaaS", + "job_ids": [ + 83, + 85, + 84, + 86, + 87, + 88, + 89 + ], + "job_statuses": [ + "finished", + "finished", + "finished", + "failed", + "aborted", + "finished", + "finished" + ], + "off_with_endpoints_no_export": true, + "independent_execution_trace_with_submission_link": false, + "off_submit_on_execute_no_link": true, + "on_submit_off_execute_carrier_preserved": true, + "postgresql_log_rows": 7, + "legacy_job_trace_ids_preserved": true, + "carrier_capacity_rejected": [ + "submission_traceparent", + "submission_tracestate" + ], + "http_requests": 3, + "http_client_spans": 0, + "http_server_spans": 0, + "http_parent_child_propagation": false, + "http_errors_not_retried": true, + "canary_absent_at_first_otlp_export": true, + "outage_job_finished": true, + "outage_shutdown_seconds": 0.0009074578993022442, + "trace_spans": 0, + "otlp_requests": 1 +} diff --git a/tasks/observability-foundation/experiments/evidence/runtime-probe.json b/tasks/observability-foundation/experiments/evidence/runtime-probe.json new file mode 100644 index 0000000..1302a6a --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/runtime-probe.json @@ -0,0 +1,68 @@ +{ + "run_id": "069a7179b169419197b1261197e6f954", + "signal_mode": "traces-logs-metrics", + "operation_counts": { + "job.submit:success": 4, + "job.execute:success": 2, + "job.execute:error": 1, + "job.execute:cancelled": 1, + "agent.tool:error": 1, + "agent.tool:success": 1, + "http.server:success": 1, + "http.server:error": 2, + "peer.http:success": 1, + "peer.http:error": 2 + }, + "duration_sample_counts": { + "job.submit:success": 4, + "job.execute:success": 2, + "job.execute:error": 1, + "job.execute:cancelled": 1, + "agent.tool:error": 1, + "agent.tool:success": 1, + "http.server:success": 1, + "http.server:error": 2, + "peer.http:success": 1, + "peer.http:error": 2 + }, + "concurrent_tool_outcomes_isolated": true, + "scope": "disposable PostgreSQL + production Job/PG log/Peer HTTP/ASGI mechanisms; not full runtime or SaaS", + "job_ids": [ + 76, + 78, + 77, + 79, + 80, + 81, + 82 + ], + "job_statuses": [ + "finished", + "finished", + "finished", + "failed", + "aborted", + "finished", + "finished" + ], + "off_with_endpoints_no_export": true, + "independent_execution_trace_with_submission_link": true, + "off_submit_on_execute_no_link": true, + "on_submit_off_execute_carrier_preserved": true, + "postgresql_log_rows": 7, + "legacy_job_trace_ids_preserved": true, + "carrier_capacity_rejected": [ + "submission_traceparent", + "submission_tracestate" + ], + "http_requests": 3, + "http_client_spans": 3, + "http_server_spans": 3, + "http_parent_child_propagation": true, + "http_errors_not_retried": true, + "canary_absent_at_first_otlp_export": true, + "outage_job_finished": true, + "outage_shutdown_seconds": 0.8620116249658167, + "trace_spans": 16, + "otlp_requests": 3 +} diff --git a/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/cloudflare-free-check.json b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/cloudflare-free-check.json new file mode 100644 index 0000000..bba1c8a --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/cloudflare-free-check.json @@ -0,0 +1,10 @@ +{ + "date": "2026-10-03", + "scope": "Cloudflare official-document research and zero-service-fee packet revision only", + "markdown_links_fences_whitespace": "pass", + "markdown_files_checked": 13, + "tracked_source_unchanged": true, + "cloud_account_runtime_validation": "not run", + "paid_services_enabled": false, + "experiments_started_this_turn": false +} diff --git a/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/opt-in-packet-check.json b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/opt-in-packet-check.json new file mode 100644 index 0000000..cb54bb2 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/opt-in-packet-check.json @@ -0,0 +1,23 @@ +{ + "checked_at": "2026-10-03T08:38:27.879502+00:00", + "scope": "D10 accepted Grafana Cloud Free, vendor-neutral opt-in and existing PostgreSQL log retention; packet only", + "markdown_files_checked": 13, + "local_markdown_links_checked": 136, + "markdown_links_fences_whitespace": "pass", + "hub_patch_apply_check": "pass", + "hub_patch_temporary_application_links_and_svc_block": "pass", + "hub_patch_files": [ + "20-product-tdd/observability-contract.md", + "docs/index.md", + "20-product-tdd/cross-unit-contracts.md", + "20-product-tdd/knowledge-capability-contract.md" + ], + "core_head": "1385e6066336eee37626e3b4f4bf68155f9bf1b7", + "hub_head": "42f7bad1c61e57b5e0ebf55e27815ddc2ae913fa", + "tracked_source_unchanged": true, + "hub_and_shared_worktrees_clean": true, + "application_runtime_validation": "not run; V0/V3 remain pending implementation", + "cloud_account_runtime_validation": "not run", + "paid_services_enabled": false, + "experiments_started_this_turn": false +} diff --git a/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/packet-check.json b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/packet-check.json new file mode 100644 index 0000000..d36ce43 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/packet-check.json @@ -0,0 +1,23 @@ +{ + "date": "2026-10-03", + "counts": { + "markdown": 13, + "python": 15, + "javascript": 3, + "json": 25, + "jsonl": 3 + }, + "local_links_fences_whitespace": "pass", + "hub_patch_apply_and_generated_regions": "pass", + "hub_patch_files": [ + "20-product-tdd/observability-contract.md", + "docs/index.md", + "20-product-tdd/cross-unit-contracts.md", + "20-product-tdd/knowledge-capability-contract.md" + ], + "tracked_source_hub_shared_unchanged": true, + "sentinel_archived_assertions": "pass", + "cloud_account_validation": "not run", + "resource_check": "resource-final.json", + "full_repository_tests": "not applicable: task packet and isolated experiment only" +} diff --git a/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/resource-final.json b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/resource-final.json new file mode 100644 index 0000000..76d6888 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/resource-final.json @@ -0,0 +1,55 @@ +{ + "containers": [ + { + "name": "inkcre-o11y-g1-b0a97f7c-openobserve-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-g1-b0a97f7c-collector-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-tempo-g1-b0a97f7c-tempo-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-tempo-g1-b0a97f7c-collector-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-stack-g1-b0a97f7c-grafana-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-stack-g1-b0a97f7c-prometheus-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-stack-g1-b0a97f7c-tempo-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-stack-g1-b0a97f7c-loki-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-stack-g1-b0a97f7c-collector-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-db-g1-b0a97f7c-postgrest-1", + "state": "exited" + }, + { + "name": "inkcre-o11y-db-g1-b0a97f7c-postgres-1", + "state": "exited" + } + ], + "sockets_present": { + "inkcre-o11y-g1-b0a97f7c": false, + "inkcre-o11y-tempo-g1-b0a97f7c": false, + "inkcre-o11y-stack-g1-b0a97f7c": false, + "inkcre-o11y-db-g1-b0a97f7c": false + }, + "scope": "Only four synthetic experiment projects; volumes preserved; SVC database untouched" +} diff --git a/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel-input.json b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel-input.json new file mode 100644 index 0000000..85beffd --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel-input.json @@ -0,0 +1,322 @@ +{ + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.name", + "value": { + "stringValue": "inkcre-sentinel-probe" + } + } + ] + }, + "scopeSpans": [ + { + "spans": [ + { + "name": "sentinel", + "traceId": "1b04f2d5ac184a60b9d319552bb16762", + "spanId": "31fd0a1179c64926", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": -1.0 + } + } + ] + }, + { + "name": "partial", + "traceId": "fd653d5fc1124a909e437f42f5006e60", + "spanId": "4b600e435ad644a8", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": -1.0 + } + } + ] + }, + { + "name": "zero", + "traceId": "2a2c264014af427187ad672681fb37e9", + "spanId": "85e1db950c3a49dc", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.0 + } + } + ] + }, + { + "name": "known", + "traceId": "3b2b183fe7e3459b903f029d56d32429", + "spanId": "888307fc231f4c57", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.125 + } + } + ] + }, + { + "name": "usage_only_sentinel", + "traceId": "df75889bbc7a48639bbb1dee6deaa80b", + "spanId": "d51d0c5b7c8e49b8", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "-1" + } + } + ] + }, + { + "name": "source_status", + "traceId": "36cf13bf12924978bce92114b8fa5e9c", + "spanId": "b2b49af9e5254697", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "inkcre.ai.usage.input.source", + "value": { + "stringValue": "unavailable" + } + }, + { + "key": "inkcre.ai.usage.output.source", + "value": { + "stringValue": "unavailable" + } + }, + { + "key": "inkcre.ai.cost.source", + "value": { + "stringValue": "unavailable" + } + } + ] + } + ] + } + ] + } + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel.json b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel.json new file mode 100644 index 0000000..faad482 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/sentinel-saas-20261003/sentinel.json @@ -0,0 +1,605 @@ +{ + "input": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.name", + "value": { + "stringValue": "inkcre-sentinel-probe" + } + } + ] + }, + "scopeSpans": [ + { + "spans": [ + { + "name": "sentinel", + "traceId": "1b04f2d5ac184a60b9d319552bb16762", + "spanId": "31fd0a1179c64926", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": -1.0 + } + } + ] + }, + { + "name": "partial", + "traceId": "fd653d5fc1124a909e437f42f5006e60", + "spanId": "4b600e435ad644a8", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": -1.0 + } + } + ] + }, + { + "name": "zero", + "traceId": "2a2c264014af427187ad672681fb37e9", + "spanId": "85e1db950c3a49dc", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.0 + } + } + ] + }, + { + "name": "known", + "traceId": "3b2b183fe7e3459b903f029d56d32429", + "spanId": "888307fc231f4c57", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.125 + } + } + ] + }, + { + "name": "usage_only_sentinel", + "traceId": "df75889bbc7a48639bbb1dee6deaa80b", + "spanId": "d51d0c5b7c8e49b8", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "-1" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "-1" + } + } + ] + }, + { + "name": "source_status", + "traceId": "36cf13bf12924978bce92114b8fa5e9c", + "spanId": "b2b49af9e5254697", + "startTimeUnixNano": "1791011243272046000", + "endTimeUnixNano": "1791011243272047000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "c39535ef9e7343a3912f1a6c931b5486" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.request.model", + "value": { + "stringValue": "gpt-4o" + } + }, + { + "key": "inkcre.ai.usage.input.source", + "value": { + "stringValue": "unavailable" + } + }, + { + "key": "inkcre.ai.usage.output.source", + "value": { + "stringValue": "unavailable" + } + }, + { + "key": "inkcre.ai.cost.source", + "value": { + "stringValue": "unavailable" + } + } + ] + } + ] + } + ] + } + ] + }, + "readback": { + "took": 8, + "took_detail": { + "total": 8, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 7 + }, + "hits": [ + { + "_o2_ingest_ts": 1791011244507094, + "_timestamp": 1791011243272046, + "duration": 1, + "end_time": 1791011243272047000, + "events": "[]", + "flags": 1, + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_request_model": "gpt-4o", + "gen_ai_response_model": "gpt-4o", + "gen_ai_usage_cost": -1.0, + "gen_ai_usage_cost_input": 2e-05, + "gen_ai_usage_input_tokens": 8, + "gen_ai_usage_output_tokens": -1, + "gen_ai_usage_total_tokens": 8, + "inkcre_lab_run_id": "c39535ef9e7343a3912f1a6c931b5486", + "links": "[]", + "operation_name": "partial", + "service_name": "inkcre-sentinel-probe", + "span_id": "4b600e435ad644a8", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1791011243272046000, + "status_code": 0, + "status_message": "", + "trace_id": "fd653d5fc1124a909e437f42f5006e60" + }, + { + "_o2_ingest_ts": 1791011244507114, + "_timestamp": 1791011243272046, + "duration": 1, + "end_time": 1791011243272047000, + "events": "[]", + "flags": 1, + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_request_model": "gpt-4o", + "gen_ai_response_model": "gpt-4o", + "gen_ai_usage_cost": 0.0, + "gen_ai_usage_input_tokens": 0, + "gen_ai_usage_output_tokens": 0, + "gen_ai_usage_total_tokens": 0, + "inkcre_lab_run_id": "c39535ef9e7343a3912f1a6c931b5486", + "links": "[]", + "operation_name": "zero", + "service_name": "inkcre-sentinel-probe", + "span_id": "85e1db950c3a49dc", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1791011243272046000, + "status_code": 0, + "status_message": "", + "trace_id": "2a2c264014af427187ad672681fb37e9" + }, + { + "_o2_ingest_ts": 1791011244507131, + "_timestamp": 1791011243272046, + "duration": 1, + "end_time": 1791011243272047000, + "events": "[]", + "flags": 1, + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_request_model": "gpt-4o", + "gen_ai_response_model": "gpt-4o", + "gen_ai_usage_cost": 0.125, + "gen_ai_usage_input_tokens": 8, + "gen_ai_usage_output_tokens": 3, + "gen_ai_usage_total_tokens": 11, + "inkcre_lab_run_id": "c39535ef9e7343a3912f1a6c931b5486", + "links": "[]", + "operation_name": "known", + "service_name": "inkcre-sentinel-probe", + "span_id": "888307fc231f4c57", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1791011243272046000, + "status_code": 0, + "status_message": "", + "trace_id": "3b2b183fe7e3459b903f029d56d32429" + }, + { + "_o2_ingest_ts": 1791011244507148, + "_timestamp": 1791011243272046, + "duration": 1, + "end_time": 1791011243272047000, + "events": "[]", + "flags": 1, + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_request_model": "gpt-4o", + "gen_ai_response_model": "gpt-4o", + "gen_ai_usage_cost": -0.0, + "gen_ai_usage_input_tokens": -1, + "gen_ai_usage_output_tokens": -1, + "gen_ai_usage_total_tokens": 0, + "inkcre_lab_run_id": "c39535ef9e7343a3912f1a6c931b5486", + "links": "[]", + "operation_name": "usage_only_sentinel", + "service_name": "inkcre-sentinel-probe", + "span_id": "d51d0c5b7c8e49b8", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1791011243272046000, + "status_code": 0, + "status_message": "", + "trace_id": "df75889bbc7a48639bbb1dee6deaa80b" + }, + { + "_o2_ingest_ts": 1791011244507167, + "_timestamp": 1791011243272046, + "duration": 1, + "end_time": 1791011243272047000, + "events": "[]", + "flags": 1, + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_request_model": "gpt-4o", + "gen_ai_response_model": "gpt-4o", + "gen_ai_usage_cost": -0.0, + "gen_ai_usage_input_tokens": 0, + "gen_ai_usage_output_tokens": 0, + "gen_ai_usage_total_tokens": 0, + "inkcre_ai_cost_source": "unavailable", + "inkcre_ai_usage_input_source": "unavailable", + "inkcre_ai_usage_output_source": "unavailable", + "inkcre_lab_run_id": "c39535ef9e7343a3912f1a6c931b5486", + "links": "[]", + "operation_name": "source_status", + "service_name": "inkcre-sentinel-probe", + "span_id": "b2b49af9e5254697", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1791011243272046000, + "status_code": 0, + "status_message": "", + "trace_id": "36cf13bf12924978bce92114b8fa5e9c" + }, + { + "_o2_ingest_ts": 1791011244507068, + "_timestamp": 1791011243272046, + "duration": 1, + "end_time": 1791011243272047000, + "events": "[]", + "flags": 1, + "gen_ai_operation_name": "chat", + "gen_ai_provider_name": "synthetic", + "gen_ai_request_model": "gpt-4o", + "gen_ai_response_model": "gpt-4o", + "gen_ai_usage_cost": -1.0, + "gen_ai_usage_input_tokens": -1, + "gen_ai_usage_output_tokens": -1, + "gen_ai_usage_total_tokens": 0, + "inkcre_lab_run_id": "c39535ef9e7343a3912f1a6c931b5486", + "links": "[]", + "operation_name": "sentinel", + "service_name": "inkcre-sentinel-probe", + "span_id": "31fd0a1179c64926", + "span_kind": "0", + "span_status": "UNSET", + "start_time": 1791011243272046000, + "status_code": 0, + "status_message": "", + "trace_id": "1b04f2d5ac184a60b9d319552bb16762" + } + ], + "total": 6, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 18, + "trace_id": "01a10096cb0d7a81914240e3c990ac27", + "is_partial": false, + "result_cache_ratio": 0, + "order_by": "desc", + "order_by_metadata": [ + [ + "_timestamp", + "desc" + ] + ], + "is_histogram_eligible": true, + "peak_memory_usage": 6.0 + }, + "duplicate_rows": 0, + "summary": { + "partial": { + "gen_ai_usage_input_tokens": 8, + "gen_ai_usage_output_tokens": -1, + "gen_ai_usage_total_tokens": 8, + "gen_ai_usage_cost": -1.0, + "inkcre_ai_usage_input_source": null, + "inkcre_ai_cost_source": null + }, + "zero": { + "gen_ai_usage_input_tokens": 0, + "gen_ai_usage_output_tokens": 0, + "gen_ai_usage_total_tokens": 0, + "gen_ai_usage_cost": 0.0, + "inkcre_ai_usage_input_source": null, + "inkcre_ai_cost_source": null + }, + "known": { + "gen_ai_usage_input_tokens": 8, + "gen_ai_usage_output_tokens": 3, + "gen_ai_usage_total_tokens": 11, + "gen_ai_usage_cost": 0.125, + "inkcre_ai_usage_input_source": null, + "inkcre_ai_cost_source": null + }, + "usage_only_sentinel": { + "gen_ai_usage_input_tokens": -1, + "gen_ai_usage_output_tokens": -1, + "gen_ai_usage_total_tokens": 0, + "gen_ai_usage_cost": -0.0, + "inkcre_ai_usage_input_source": null, + "inkcre_ai_cost_source": null + }, + "source_status": { + "gen_ai_usage_input_tokens": 0, + "gen_ai_usage_output_tokens": 0, + "gen_ai_usage_total_tokens": 0, + "gen_ai_usage_cost": -0.0, + "inkcre_ai_usage_input_source": "unavailable", + "inkcre_ai_cost_source": "unavailable" + }, + "sentinel": { + "gen_ai_usage_input_tokens": -1, + "gen_ai_usage_output_tokens": -1, + "gen_ai_usage_total_tokens": 0, + "gen_ai_usage_cost": -1.0, + "inkcre_ai_usage_input_source": null, + "inkcre_ai_cost_source": null + } + }, + "aggregate_readback": { + "took": 47, + "took_detail": { + "total": 47, + "cache_took": 0, + "file_list_took": 0, + "wait_in_queue": 0, + "idx_took": 0, + "search_took": 45 + }, + "hits": [ + { + "naive_input": 7, + "naive_cost": -0.875, + "known_input": 8, + "known_cost": 0.125 + } + ], + "total": 1, + "from": 0, + "size": 100, + "cached_ratio": 0, + "scan_size": 0, + "idx_scan_size": 0, + "scan_records": 18, + "trace_id": "01a10096cb1b744390707011c985c4f6", + "is_partial": false, + "result_cache_ratio": 0, + "is_histogram_eligible": true, + "peak_memory_usage": 0.0 + }, + "scope": "Synthetic fixed-version backend behavior only; not a production mapping or Cloud test" +} diff --git a/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-after.json b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-after.json new file mode 100644 index 0000000..d3955ce --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-after.json @@ -0,0 +1,367 @@ +{ + "submission": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "name": "submission", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "48356" + } + }, + "absent": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "dARq3wcsT7mPAP+KRSBvng==", + "spanId": "ijFiY4dIRpU=", + "name": "absent", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "traceState": "inkcre=synthetic", + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "48356" + } + }, + "explicit": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "j10OVyrHQA+LmnSmQiFWTQ==", + "spanId": "8KvJ5ZHaTfo=", + "name": "explicit", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.agent.version", + "value": { + "stringValue": "agent-7" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.125 + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "traceState": "inkcre=synthetic", + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "48356" + } + }, + "zero": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "hw2CNWxnRWuXGpjYI5Xt3g==", + "spanId": "+PSV3VV5T3U=", + "name": "zero", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0 + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "traceState": "inkcre=synthetic", + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "48356" + } + } +} diff --git a/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-before.json b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-before.json new file mode 100644 index 0000000..123abf2 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-before.json @@ -0,0 +1,367 @@ +{ + "submission": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "name": "submission", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "24178" + } + }, + "absent": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "dARq3wcsT7mPAP+KRSBvng==", + "spanId": "ijFiY4dIRpU=", + "name": "absent", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "traceState": "inkcre=synthetic", + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "24178" + } + }, + "explicit": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "j10OVyrHQA+LmnSmQiFWTQ==", + "spanId": "8KvJ5ZHaTfo=", + "name": "explicit", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.agent.version", + "value": { + "stringValue": "agent-7" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.125 + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "traceState": "inkcre=synthetic", + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "24178" + } + }, + "zero": { + "trace": { + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + }, + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + } + ] + }, + "scopeSpans": [ + { + "scope": {}, + "spans": [ + { + "traceId": "hw2CNWxnRWuXGpjYI5Xt3g==", + "spanId": "+PSV3VV5T3U=", + "name": "zero", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0 + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "mc9PBUcQRCCX7BuDCAntIg==", + "spanId": "tIQuR9o2Sao=", + "traceState": "inkcre=synthetic", + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ], + "status": {} + } + ] + } + ] + } + ] + }, + "metrics": { + "inspectedBytes": "24178" + } + } +} diff --git a/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-input.json b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-input.json new file mode 100644 index 0000000..14848f1 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-input.json @@ -0,0 +1,256 @@ +{ + "resourceSpans": [ + { + "resource": { + "attributes": [ + { + "key": "service.name", + "value": { + "stringValue": "inkcre-projection-probe" + } + }, + { + "key": "service.version", + "value": { + "stringValue": "service-3" + } + } + ] + }, + "scopeSpans": [ + { + "spans": [ + { + "name": "submission", + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + }, + { + "name": "absent", + "traceId": "74046adf072c4fb98f00ff8a45206f9e", + "spanId": "8a31626387484695", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "traceState": "inkcre=synthetic", + "flags": 1, + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ] + }, + { + "name": "explicit", + "traceId": "8f5d0e572ac7400f8b9a74a64221564d", + "spanId": "f0abc9e591da4dfa", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.agent.version", + "value": { + "stringValue": "agent-7" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "8" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "3" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.125 + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "traceState": "inkcre=synthetic", + "flags": 1, + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ] + }, + { + "name": "zero", + "traceId": "870d82356c67456b971a98d82395edde", + "spanId": "f8f495dd55794f75", + "startTimeUnixNano": "1791004150502892000", + "endTimeUnixNano": "1791004150502893000", + "attributes": [ + { + "key": "inkcre.lab.run_id", + "value": { + "stringValue": "20d50f4624f645d8bf2ee385473de7f0" + } + }, + { + "key": "gen_ai.operation.name", + "value": { + "stringValue": "chat" + } + }, + { + "key": "gen_ai.provider.name", + "value": { + "stringValue": "synthetic" + } + }, + { + "key": "gen_ai.usage.input_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.output_tokens", + "value": { + "intValue": "0" + } + }, + { + "key": "gen_ai.usage.cost", + "value": { + "doubleValue": 0.0 + } + }, + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ], + "links": [ + { + "traceId": "99cf4f054710442097ec1b830809ed22", + "spanId": "b4842e47da3649aa", + "traceState": "inkcre=synthetic", + "flags": 1, + "attributes": [ + { + "key": "inkcre.link.reason", + "value": { + "stringValue": "job-submission" + } + }, + { + "key": "inkcre.link.sequence", + "value": { + "intValue": "7" + } + } + ] + } + ] + } + ] + } + ] + } + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-metrics.txt b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-metrics.txt new file mode 100644 index 0000000..73478c3 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-metrics.txt @@ -0,0 +1,1477 @@ +# HELP deprecated_flags_inuse_total The number of deprecated flags currently set. +# TYPE deprecated_flags_inuse_total counter +deprecated_flags_inuse_total 0 +# HELP go_gc_duration_seconds A summary of the wall-time pause (stop-the-world) duration in garbage collection cycles. +# TYPE go_gc_duration_seconds summary +go_gc_duration_seconds{quantile="0"} 6.0303e-05 +go_gc_duration_seconds{quantile="0.25"} 7.3604e-05 +go_gc_duration_seconds{quantile="0.5"} 0.000116607 +go_gc_duration_seconds{quantile="0.75"} 0.000151613 +go_gc_duration_seconds{quantile="1"} 0.00274444 +go_gc_duration_seconds_sum 0.204987504 +go_gc_duration_seconds_count 1180 +# HELP go_gc_gogc_percent Heap size target percentage configured by the user, otherwise 100. This value is set by the GOGC environment variable, and the runtime/debug.SetGCPercent function. Sourced from /gc/gogc:percent. +# TYPE go_gc_gogc_percent gauge +go_gc_gogc_percent 100 +# HELP go_gc_gomemlimit_bytes Go runtime memory limit configured by the user, otherwise math.MaxInt64. This value is set by the GOMEMLIMIT environment variable, and the runtime/debug.SetMemoryLimit function. Sourced from /gc/gomemlimit:bytes. +# TYPE go_gc_gomemlimit_bytes gauge +go_gc_gomemlimit_bytes 9.223372036854776e+18 +# HELP go_goroutines Number of goroutines that currently exist. +# TYPE go_goroutines gauge +go_goroutines 460 +# HELP go_info Information about the Go environment. +# TYPE go_info gauge +go_info{version="go1.27.1"} 1 +# HELP go_memstats_alloc_bytes Number of bytes allocated in heap and currently in use. Equals to /memory/classes/heap/objects:bytes. +# TYPE go_memstats_alloc_bytes gauge +go_memstats_alloc_bytes 1.21627344e+08 +# HELP go_memstats_alloc_bytes_total Total number of bytes allocated in heap until now, even if released already. Equals to /gc/heap/allocs:bytes. +# TYPE go_memstats_alloc_bytes_total counter +go_memstats_alloc_bytes_total 7.307234008e+09 +# HELP go_memstats_buck_hash_sys_bytes Number of bytes used by the profiling bucket hash table. Equals to /memory/classes/profiling/buckets:bytes. +# TYPE go_memstats_buck_hash_sys_bytes gauge +go_memstats_buck_hash_sys_bytes 1.58975e+06 +# HELP go_memstats_frees_total Total number of heap objects frees. Equals to /gc/heap/frees:objects + /gc/heap/tiny/allocs:objects. +# TYPE go_memstats_frees_total counter +go_memstats_frees_total 8.1092164e+07 +# HELP go_memstats_gc_sys_bytes Number of bytes used for garbage collection system metadata. Equals to /memory/classes/metadata/other:bytes. +# TYPE go_memstats_gc_sys_bytes gauge +go_memstats_gc_sys_bytes 4.18536e+06 +# HELP go_memstats_heap_alloc_bytes Number of heap bytes allocated and currently in use, same as go_memstats_alloc_bytes. Equals to /memory/classes/heap/objects:bytes. +# TYPE go_memstats_heap_alloc_bytes gauge +go_memstats_heap_alloc_bytes 1.21627344e+08 +# HELP go_memstats_heap_idle_bytes Number of heap bytes waiting to be used. Equals to /memory/classes/heap/released:bytes + /memory/classes/heap/free:bytes. +# TYPE go_memstats_heap_idle_bytes gauge +go_memstats_heap_idle_bytes 5.865472e+06 +# HELP go_memstats_heap_inuse_bytes Number of heap bytes that are in use. Equals to /memory/classes/heap/objects:bytes + /memory/classes/heap/unused:bytes +# TYPE go_memstats_heap_inuse_bytes gauge +go_memstats_heap_inuse_bytes 1.24551168e+08 +# HELP go_memstats_heap_objects Number of currently allocated objects. Equals to /gc/heap/objects:objects. +# TYPE go_memstats_heap_objects gauge +go_memstats_heap_objects 102439 +# HELP go_memstats_heap_released_bytes Number of heap bytes released to OS. Equals to /memory/classes/heap/released:bytes. +# TYPE go_memstats_heap_released_bytes gauge +go_memstats_heap_released_bytes 1.531904e+06 +# HELP go_memstats_heap_sys_bytes Number of heap bytes obtained from system. Equals to /memory/classes/heap/objects:bytes + /memory/classes/heap/unused:bytes + /memory/classes/heap/released:bytes + /memory/classes/heap/free:bytes. +# TYPE go_memstats_heap_sys_bytes gauge +go_memstats_heap_sys_bytes 1.3041664e+08 +# HELP go_memstats_last_gc_time_seconds Number of seconds since 1970 of last garbage collection. +# TYPE go_memstats_last_gc_time_seconds gauge +go_memstats_last_gc_time_seconds 1.7910043131303544e+09 +# HELP go_memstats_mallocs_total Total number of heap objects allocated, both live and gc-ed. Semantically a counter version for go_memstats_heap_objects gauge. Equals to /gc/heap/allocs:objects + /gc/heap/tiny/allocs:objects. +# TYPE go_memstats_mallocs_total counter +go_memstats_mallocs_total 8.1194603e+07 +# HELP go_memstats_mcache_inuse_bytes Number of bytes in use by mcache structures. Equals to /memory/classes/metadata/mcache/inuse:bytes. +# TYPE go_memstats_mcache_inuse_bytes gauge +go_memstats_mcache_inuse_bytes 4592 +# HELP go_memstats_mcache_sys_bytes Number of bytes used for mcache structures obtained from system. Equals to /memory/classes/metadata/mcache/inuse:bytes + /memory/classes/metadata/mcache/free:bytes. +# TYPE go_memstats_mcache_sys_bytes gauge +go_memstats_mcache_sys_bytes 16072 +# HELP go_memstats_mspan_inuse_bytes Number of bytes in use by mspan structures. Equals to /memory/classes/metadata/mspan/inuse:bytes. +# TYPE go_memstats_mspan_inuse_bytes gauge +go_memstats_mspan_inuse_bytes 369440 +# HELP go_memstats_mspan_sys_bytes Number of bytes used for mspan structures obtained from system. Equals to /memory/classes/metadata/mspan/inuse:bytes + /memory/classes/metadata/mspan/free:bytes. +# TYPE go_memstats_mspan_sys_bytes gauge +go_memstats_mspan_sys_bytes 391680 +# HELP go_memstats_next_gc_bytes Number of heap bytes when next garbage collection will take place. Equals to /gc/heap/goal:bytes. +# TYPE go_memstats_next_gc_bytes gauge +go_memstats_next_gc_bytes 2.23517778e+08 +# HELP go_memstats_other_sys_bytes Number of bytes used for other system allocations. Equals to /memory/classes/other:bytes. +# TYPE go_memstats_other_sys_bytes gauge +go_memstats_other_sys_bytes 876890 +# HELP go_memstats_stack_inuse_bytes Number of bytes obtained from system for stack allocator in non-CGO environments. Equals to /memory/classes/heap/stacks:bytes. +# TYPE go_memstats_stack_inuse_bytes gauge +go_memstats_stack_inuse_bytes 3.801088e+06 +# HELP go_memstats_stack_sys_bytes Number of bytes obtained from system for stack allocator. Equals to /memory/classes/heap/stacks:bytes + /memory/classes/os-stacks:bytes. +# TYPE go_memstats_stack_sys_bytes gauge +go_memstats_stack_sys_bytes 3.801088e+06 +# HELP go_memstats_sys_bytes Number of bytes obtained from system. Equals to /memory/classes/total:byte. +# TYPE go_memstats_sys_bytes gauge +go_memstats_sys_bytes 1.4127748e+08 +# HELP go_sched_gomaxprocs_threads The current runtime.GOMAXPROCS setting, or the number of operating system threads that can execute user-level Go code simultaneously. Sourced from /sched/gomaxprocs:threads. +# TYPE go_sched_gomaxprocs_threads gauge +go_sched_gomaxprocs_threads 2 +# HELP go_threads Number of OS threads created. +# TYPE go_threads gauge +go_threads 8 +# HELP grpc_concurrent_streams_by_conn_max The current number of concurrent streams in the connection with the most concurrent streams. +# TYPE grpc_concurrent_streams_by_conn_max gauge +grpc_concurrent_streams_by_conn_max 0 +# HELP kv_request_duration_seconds Time spent on kv store requests. +# TYPE kv_request_duration_seconds histogram +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.005"} 56236 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.01"} 56236 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.025"} 56236 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.05"} 56237 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.1"} 56241 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.25"} 56244 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.5"} 56244 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="1"} 56244 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="2.5"} 56244 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="5"} 56244 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="10"} 56244 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory",le="+Inf"} 56244 +kv_request_duration_seconds_sum{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory"} 3.7237737100000183 +kv_request_duration_seconds_count{kv_name="livestore-partitions-lifecycler",operation="CAS",role="primary",status_code="200",type="inmemory"} 56244 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="0.005"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="0.01"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="0.025"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="0.05"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="0.1"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="0.25"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="0.5"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="1"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="2.5"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="5"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="10"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory",le="+Inf"} 1 +kv_request_duration_seconds_sum{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory"} 1.0901e-05 +kv_request_duration_seconds_count{kv_name="livestore-partitions-watcher",operation="GET",role="primary",status_code="200",type="inmemory"} 1 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.005"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.01"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.025"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.05"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.1"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.25"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.5"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="1"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="2.5"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="5"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="10"} 0 +kv_request_duration_seconds_bucket{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="+Inf"} 1 +kv_request_duration_seconds_sum{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory"} 140604.862280857 +kv_request_duration_seconds_count{kv_name="livestore-partitions-watcher",operation="WatchKey",role="primary",status_code="200",type="inmemory"} 1 +# HELP process_cpu_seconds_total Total user and system CPU time spent in seconds. +# TYPE process_cpu_seconds_total counter +process_cpu_seconds_total 380.21 +# HELP process_max_fds Maximum number of open file descriptors. +# TYPE process_max_fds gauge +process_max_fds 1.048576e+06 +# HELP process_network_receive_bytes_total Number of bytes received by the process over the network. +# TYPE process_network_receive_bytes_total counter +process_network_receive_bytes_total 4.511686e+06 +# HELP process_network_transmit_bytes_total Number of bytes sent by the process over the network. +# TYPE process_network_transmit_bytes_total counter +process_network_transmit_bytes_total 4.712863e+06 +# HELP process_open_fds Number of open file descriptors. +# TYPE process_open_fds gauge +process_open_fds 16 +# HELP process_resident_memory_bytes Resident memory size in bytes. +# TYPE process_resident_memory_bytes gauge +process_resident_memory_bytes 8.4594688e+07 +# HELP process_start_time_seconds Start time of the process since unix epoch in seconds. +# TYPE process_start_time_seconds gauge +process_start_time_seconds 1.79086371864e+09 +# HELP process_virtual_memory_bytes Virtual memory size in bytes. +# TYPE process_virtual_memory_bytes gauge +process_virtual_memory_bytes 1.541992448e+09 +# HELP process_virtual_memory_max_bytes Maximum amount of virtual memory available in bytes. +# TYPE process_virtual_memory_max_bytes gauge +process_virtual_memory_max_bytes 1.8446744073709552e+19 +# HELP prometheus_remote_storage_exemplars_in_total Exemplars in to remote storage, compare to exemplars out for queue managers. Deprecated, check prometheus_wal_watcher_records_read_total and prometheus_remote_storage_exemplars_dropped_total +# TYPE prometheus_remote_storage_exemplars_in_total counter +prometheus_remote_storage_exemplars_in_total 0 +# HELP prometheus_remote_storage_histograms_in_total HistogramSamples in to remote storage, compare to histograms out for queue managers. Deprecated, check prometheus_wal_watcher_records_read_total and prometheus_remote_storage_histograms_dropped_total +# TYPE prometheus_remote_storage_histograms_in_total counter +prometheus_remote_storage_histograms_in_total 0 +# HELP prometheus_remote_storage_samples_in_total Samples in to remote storage, compare to samples out for queue managers. Deprecated, check prometheus_wal_watcher_records_read_total and prometheus_remote_storage_samples_dropped_total +# TYPE prometheus_remote_storage_samples_in_total counter +prometheus_remote_storage_samples_in_total 0 +# HELP prometheus_remote_storage_string_interner_zero_reference_releases_total The number of times release has been called for strings that are not interned. +# TYPE prometheus_remote_storage_string_interner_zero_reference_releases_total counter +prometheus_remote_storage_string_interner_zero_reference_releases_total 0 +# HELP prometheus_template_text_expansion_failures_total The total number of template text expansion failures. +# TYPE prometheus_template_text_expansion_failures_total counter +prometheus_template_text_expansion_failures_total 0 +# HELP prometheus_template_text_expansions_total The total number of template text expansions. +# TYPE prometheus_template_text_expansions_total counter +prometheus_template_text_expansions_total 0 +# HELP tempo_backend_scheduler_compaction_empty_tenant_cycle_total The number of compaction cycles where no tenant had work available +# TYPE tempo_backend_scheduler_compaction_empty_tenant_cycle_total counter +tempo_backend_scheduler_compaction_empty_tenant_cycle_total 469 +# HELP tempo_backend_scheduler_compaction_tenant_empty_job_total The number of times an empty job was received from the priority queue +# TYPE tempo_backend_scheduler_compaction_tenant_empty_job_total counter +tempo_backend_scheduler_compaction_tenant_empty_job_total 0 +# HELP tempo_backend_scheduler_jobs_not_found_total The number of calls to get a job that were not found +# TYPE tempo_backend_scheduler_jobs_not_found_total counter +tempo_backend_scheduler_jobs_not_found_total{worker_id="b200b42280e8"} 1984 +# HELP tempo_backend_scheduler_work_cache_file_size_bytes Size of the work cache file in bytes +# TYPE tempo_backend_scheduler_work_cache_file_size_bytes histogram +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="1024"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="2048"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="4096"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="8192"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="16384"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="32768"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="65536"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="131072"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="262144"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="524288"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="1.048576e+06"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="2.097152e+06"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="4.194304e+06"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="8.388608e+06"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="1.6777216e+07"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="3.3554432e+07"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_bucket{le="+Inf"} 0 +tempo_backend_scheduler_work_cache_file_size_bytes_sum 0 +tempo_backend_scheduler_work_cache_file_size_bytes_count 0 +# HELP tempo_backend_scheduler_work_flushes_failed_total The number of times the work cache flush to backend storage failed +# TYPE tempo_backend_scheduler_work_flushes_failed_total counter +tempo_backend_scheduler_work_flushes_failed_total 0 +# HELP tempo_backend_scheduler_work_flushes_total The number of times the work cache was flushed to backend storage +# TYPE tempo_backend_scheduler_work_flushes_total counter +tempo_backend_scheduler_work_flushes_total 2343 +# HELP tempo_backend_worker_call_retries_total Total number of retries for calls +# TYPE tempo_backend_worker_call_retries_total counter +tempo_backend_worker_call_retries_total 1984 +# HELP tempo_block_builder_consume_cycle_duration_seconds Time spent consuming a full cycle. +# TYPE tempo_block_builder_consume_cycle_duration_seconds histogram +tempo_block_builder_consume_cycle_duration_seconds_bucket{le="+Inf"} 0 +tempo_block_builder_consume_cycle_duration_seconds_sum 0 +tempo_block_builder_consume_cycle_duration_seconds_count 0 +# HELP tempo_block_builder_flush_size_bytes Size in bytes of blocks flushed by the block-builder. +# TYPE tempo_block_builder_flush_size_bytes histogram +tempo_block_builder_flush_size_bytes_bucket{le="1.048576e+06"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="2.097152e+06"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="4.194304e+06"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="8.388608e+06"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="1.6777216e+07"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="3.3554432e+07"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="6.7108864e+07"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="1.34217728e+08"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="2.68435456e+08"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="5.36870912e+08"} 0 +tempo_block_builder_flush_size_bytes_bucket{le="+Inf"} 0 +tempo_block_builder_flush_size_bytes_sum 0 +tempo_block_builder_flush_size_bytes_count 0 +# HELP tempo_buffer_pool_miss_bytes_total The total number of alloc'ed bytes that missed the sync pools. +# TYPE tempo_buffer_pool_miss_bytes_total counter +tempo_buffer_pool_miss_bytes_total{direction="over",name="cache"} 0 +tempo_buffer_pool_miss_bytes_total{direction="over",name="ingester_prealloc"} 0 +tempo_buffer_pool_miss_bytes_total{direction="under",name="cache"} 0 +tempo_buffer_pool_miss_bytes_total{direction="under",name="ingester_prealloc"} 0 +# HELP tempo_build_info A metric with a constant '1' value labeled by version, revision, branch, goversion from which tempo was built, and the goos and goarch for the build. +# TYPE tempo_build_info gauge +tempo_build_info{branch="HEAD",goarch="amd64",goos="linux",goversion="go1.27.1",revision="77fc4563d",tags="unknown",version="v3.1.0"} 1 +# HELP tempo_consul_request_duration_seconds Time spent on consul requests. +# TYPE tempo_consul_request_duration_seconds histogram +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="0.005"} 84356 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="0.01"} 84357 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="0.025"} 84357 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="0.05"} 84358 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="0.1"} 84363 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="0.25"} 84367 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="0.5"} 84367 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="1"} 84367 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="2.5"} 84367 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="5"} 84367 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="10"} 84367 +tempo_consul_request_duration_seconds_bucket{kv_name="live-store-ring",operation="CAS loop",status_code="200",le="+Inf"} 84367 +tempo_consul_request_duration_seconds_sum{kv_name="live-store-ring",operation="CAS loop",status_code="200"} 6.318466432999996 +tempo_consul_request_duration_seconds_count{kv_name="live-store-ring",operation="CAS loop",status_code="200"} 84367 +# HELP tempo_distributor_bytes_received_total The total number of proto bytes received per tenant, after limits +# TYPE tempo_distributor_bytes_received_total counter +tempo_distributor_bytes_received_total{tenant="single-tenant"} 1451 +# HELP tempo_distributor_ingress_bytes_total The total number of bytes received per tenant, before limits +# TYPE tempo_distributor_ingress_bytes_total counter +tempo_distributor_ingress_bytes_total{tenant="single-tenant"} 1451 +# HELP tempo_distributor_kafka_records_per_request The number of records in each kafka request +# TYPE tempo_distributor_kafka_records_per_request histogram +tempo_distributor_kafka_records_per_request_bucket{le="0.005"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="0.01"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="0.025"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="0.05"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="0.1"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="0.25"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="0.5"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="1"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="2.5"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="5"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="10"} 0 +tempo_distributor_kafka_records_per_request_bucket{le="+Inf"} 0 +tempo_distributor_kafka_records_per_request_sum 0 +tempo_distributor_kafka_records_per_request_count 0 +# HELP tempo_distributor_kafka_write_latency_seconds The latency of writing to kafka +# TYPE tempo_distributor_kafka_write_latency_seconds histogram +tempo_distributor_kafka_write_latency_seconds_bucket{le="0.005"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="0.01"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="0.025"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="0.05"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="0.1"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="0.25"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="0.5"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="1"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="2.5"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="5"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="10"} 0 +tempo_distributor_kafka_write_latency_seconds_bucket{le="+Inf"} 0 +tempo_distributor_kafka_write_latency_seconds_sum 0 +tempo_distributor_kafka_write_latency_seconds_count 0 +# HELP tempo_distributor_push_bytes The decoded size of each push, per tenant +# TYPE tempo_distributor_push_bytes histogram +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="1024"} 0 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="2048"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="4096"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="8192"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="16384"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="32768"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="65536"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="131072"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="262144"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="524288"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="1.048576e+06"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="2.097152e+06"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="4.194304e+06"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="8.388608e+06"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="1.6777216e+07"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="3.3554432e+07"} 1 +tempo_distributor_push_bytes_bucket{tenant="single-tenant",le="+Inf"} 1 +tempo_distributor_push_bytes_sum{tenant="single-tenant"} 1451 +tempo_distributor_push_bytes_count{tenant="single-tenant"} 1 +# HELP tempo_distributor_push_duration_seconds Records the amount of time to process and route a batch through the distributor. +# TYPE tempo_distributor_push_duration_seconds histogram +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="0.005"} 0 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="0.01"} 0 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="0.025"} 0 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="0.05"} 0 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="0.1"} 1 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="0.25"} 1 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="0.5"} 1 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="1"} 1 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="2.5"} 1 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="5"} 1 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="10"} 1 +tempo_distributor_push_duration_seconds_bucket{tenant="single-tenant",le="+Inf"} 1 +tempo_distributor_push_duration_seconds_sum{tenant="single-tenant"} 0.090444928 +tempo_distributor_push_duration_seconds_count{tenant="single-tenant"} 1 +# HELP tempo_distributor_received_traces_total The total number of traces received per tenant +# TYPE tempo_distributor_received_traces_total counter +tempo_distributor_received_traces_total{tenant="single-tenant"} 4 +# HELP tempo_distributor_spans_received_total The total number of spans received per tenant +# TYPE tempo_distributor_spans_received_total counter +tempo_distributor_spans_received_total{tenant="single-tenant"} 4 +# HELP tempo_distributor_traces_per_batch The number of traces in each batch +# TYPE tempo_distributor_traces_per_batch histogram +tempo_distributor_traces_per_batch_bucket{le="2"} 0 +tempo_distributor_traces_per_batch_bucket{le="4"} 1 +tempo_distributor_traces_per_batch_bucket{le="8"} 1 +tempo_distributor_traces_per_batch_bucket{le="16"} 1 +tempo_distributor_traces_per_batch_bucket{le="32"} 1 +tempo_distributor_traces_per_batch_bucket{le="64"} 1 +tempo_distributor_traces_per_batch_bucket{le="128"} 1 +tempo_distributor_traces_per_batch_bucket{le="256"} 1 +tempo_distributor_traces_per_batch_bucket{le="512"} 1 +tempo_distributor_traces_per_batch_bucket{le="1024"} 1 +tempo_distributor_traces_per_batch_bucket{le="+Inf"} 1 +tempo_distributor_traces_per_batch_sum 4 +tempo_distributor_traces_per_batch_count 1 +# HELP tempo_dns_failures_total The number of DNS lookup failures +# TYPE tempo_dns_failures_total counter +tempo_dns_failures_total{name="memberlist"} 0 +# HELP tempo_dns_lookups_total The number of DNS lookups resolutions attempts +# TYPE tempo_dns_lookups_total counter +tempo_dns_lookups_total{name="memberlist"} 0 +# HELP tempo_dropped_log_lines_total The total number of log lines dropped by the rate limited logger. +# TYPE tempo_dropped_log_lines_total counter +tempo_dropped_log_lines_total 0 +# HELP tempo_experimental_features_in_use_total The number of experimental features in use. +# TYPE tempo_experimental_features_in_use_total counter +tempo_experimental_features_in_use_total 0 +# HELP tempo_grpc_concurrent_streams_limit The max number of concurrent streams that can be accepted (0 means no limit). +# TYPE tempo_grpc_concurrent_streams_limit gauge +tempo_grpc_concurrent_streams_limit 100 +# HELP tempo_inflight_requests Current number of inflight requests. +# TYPE tempo_inflight_requests gauge +tempo_inflight_requests{method="GET",route="api_search"} 0 +tempo_inflight_requests{method="GET",route="api_v2_traces_traceid"} 0 +tempo_inflight_requests{method="GET",route="metrics"} 1 +tempo_inflight_requests{method="GET",route="querier_api_search"} 0 +tempo_inflight_requests{method="GET",route="querier_api_v2_traces_traceid"} 0 +tempo_inflight_requests{method="GET",route="ready"} 0 +tempo_inflight_requests{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown"} 0 +tempo_inflight_requests{method="gRPC",route="/frontend.Frontend/Process"} 0 +tempo_inflight_requests{method="gRPC",route="/grpc.health.v1.Health/Check"} 0 +tempo_inflight_requests{method="gRPC",route="/tempopb.BackendScheduler/Next"} 0 +tempo_inflight_requests{method="gRPC",route="/tempopb.Querier/FindTraceByID"} 0 +tempo_inflight_requests{method="gRPC",route="/tempopb.Querier/SearchRecent"} 0 +# HELP tempo_kv_request_duration_seconds Time spent on kv store requests. +# TYPE tempo_kv_request_duration_seconds histogram +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="0.005"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="0.01"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="0.025"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="0.05"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="0.1"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="0.25"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="0.5"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="1"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="2.5"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="5"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="10"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory",le="+Inf"} 1 +tempo_kv_request_duration_seconds_sum{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory"} 3.1602e-05 +tempo_kv_request_duration_seconds_count{kv_name="live-store-ring",operation="GET",role="primary",status_code="200",type="inmemory"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.005"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.01"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.025"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.05"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.1"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.25"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="0.5"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="1"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="2.5"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="5"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="10"} 0 +tempo_kv_request_duration_seconds_bucket{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory",le="+Inf"} 1 +tempo_kv_request_duration_seconds_sum{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory"} 140604.962872516 +tempo_kv_request_duration_seconds_count{kv_name="live-store-ring",operation="WatchKey",role="primary",status_code="200",type="inmemory"} 1 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.005"} 28115 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.01"} 28116 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.025"} 28116 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.05"} 28118 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.1"} 28120 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.25"} 28123 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="0.5"} 28123 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="1"} 28123 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="2.5"} 28123 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="5"} 28123 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="10"} 28123 +tempo_kv_request_duration_seconds_bucket{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory",le="+Inf"} 28123 +tempo_kv_request_duration_seconds_sum{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory"} 3.4263983529999944 +tempo_kv_request_duration_seconds_count{kv_name="livestore",operation="CAS",role="primary",status_code="200",type="inmemory"} 28123 +# HELP tempo_lifecycler_read_only Set to 1 if this lifecycler's instance entry is in read-only state. +# TYPE tempo_lifecycler_read_only gauge +tempo_lifecycler_read_only{name="live-store"} 0 +# HELP tempo_limits_defaults Default resource limits +# TYPE tempo_limits_defaults gauge +tempo_limits_defaults{limit_name="block_retention"} 0 +tempo_limits_defaults{limit_name="ingestion_burst_size_bytes"} 3e+07 +tempo_limits_defaults{limit_name="ingestion_rate_limit_bytes"} 3e+07 +tempo_limits_defaults{limit_name="max_blocks_per_tag_values_query"} 0 +tempo_limits_defaults{limit_name="max_bytes_per_tag_values_query"} 1e+06 +tempo_limits_defaults{limit_name="max_bytes_per_trace"} 5e+06 +tempo_limits_defaults{limit_name="max_global_traces_per_user"} 0 +tempo_limits_defaults{limit_name="max_local_traces_per_user"} 10000 +tempo_limits_defaults{limit_name="metrics_generator_max_active_series"} 0 +# HELP tempo_live_store_back_pressure_duration_seconds Duration of backpressure wait per push +# TYPE tempo_live_store_back_pressure_duration_seconds histogram +tempo_live_store_back_pressure_duration_seconds_bucket{le="0.005"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="0.01"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="0.025"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="0.05"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="0.1"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="0.25"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="0.5"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="1"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="2.5"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="5"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="10"} 0 +tempo_live_store_back_pressure_duration_seconds_bucket{le="+Inf"} 0 +tempo_live_store_back_pressure_duration_seconds_sum 0 +tempo_live_store_back_pressure_duration_seconds_count 0 +# HELP tempo_live_store_blocks_completed_total The total number of blocks completed +# TYPE tempo_live_store_blocks_completed_total counter +tempo_live_store_blocks_completed_total 1 +# HELP tempo_live_store_blocks_cut_total The total number of blocks cut by reason. +# TYPE tempo_live_store_blocks_cut_total counter +tempo_live_store_blocks_cut_total{reason="max_block_duration"} 1 +# HELP tempo_live_store_bytes_received_total The total bytes received per tenant. +# TYPE tempo_live_store_bytes_received_total counter +tempo_live_store_bytes_received_total{data_type="trace",tenant="single-tenant"} 1698 +# HELP tempo_live_store_catch_up_duration_seconds Time spent catching up at startup +# TYPE tempo_live_store_catch_up_duration_seconds gauge +tempo_live_store_catch_up_duration_seconds 0 +# HELP tempo_live_store_complete_queue_length Number of wal blocks waiting for completion +# TYPE tempo_live_store_complete_queue_length gauge +tempo_live_store_complete_queue_length 0 +# HELP tempo_live_store_completion_duration_seconds Records the amount of time to complete a block. +# TYPE tempo_live_store_completion_duration_seconds histogram +tempo_live_store_completion_duration_seconds_bucket{le="1"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="2"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="4"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="8"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="16"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="32"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="64"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="128"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="256"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="512"} 1 +tempo_live_store_completion_duration_seconds_bucket{le="+Inf"} 1 +tempo_live_store_completion_duration_seconds_sum 0.030169625 +tempo_live_store_completion_duration_seconds_count 1 +# HELP tempo_live_store_completion_failed_retries_total The total number of failed retries after a failed completion +# TYPE tempo_live_store_completion_failed_retries_total counter +tempo_live_store_completion_failed_retries_total 0 +# HELP tempo_live_store_completion_retries_total The total number of retries after a failed completion +# TYPE tempo_live_store_completion_retries_total counter +tempo_live_store_completion_retries_total 0 +# HELP tempo_live_store_completion_size_bytes Size in bytes of blocks completed. +# TYPE tempo_live_store_completion_size_bytes histogram +tempo_live_store_completion_size_bytes_bucket{le="1.048576e+06"} 1 +tempo_live_store_completion_size_bytes_bucket{le="2.097152e+06"} 1 +tempo_live_store_completion_size_bytes_bucket{le="4.194304e+06"} 1 +tempo_live_store_completion_size_bytes_bucket{le="8.388608e+06"} 1 +tempo_live_store_completion_size_bytes_bucket{le="1.6777216e+07"} 1 +tempo_live_store_completion_size_bytes_bucket{le="3.3554432e+07"} 1 +tempo_live_store_completion_size_bytes_bucket{le="6.7108864e+07"} 1 +tempo_live_store_completion_size_bytes_bucket{le="1.34217728e+08"} 1 +tempo_live_store_completion_size_bytes_bucket{le="2.68435456e+08"} 1 +tempo_live_store_completion_size_bytes_bucket{le="5.36870912e+08"} 1 +tempo_live_store_completion_size_bytes_bucket{le="+Inf"} 1 +tempo_live_store_completion_size_bytes_sum 26376 +tempo_live_store_completion_size_bytes_count 1 +# HELP tempo_live_store_failed_completions_total The total number of failed block completions +# TYPE tempo_live_store_failed_completions_total counter +tempo_live_store_failed_completions_total 0 +# HELP tempo_live_store_live_trace_bytes The current number of bytes consumed by live traces per tenant. +# TYPE tempo_live_store_live_trace_bytes gauge +tempo_live_store_live_trace_bytes{tenant="single-tenant"} 0 +# HELP tempo_live_store_live_traces The current number of live traces per tenant. +# TYPE tempo_live_store_live_traces gauge +tempo_live_store_live_traces{tenant="single-tenant"} 0 +# HELP tempo_live_store_local_blocks_flushed_total The total number of local complete blocks flushed +# TYPE tempo_live_store_local_blocks_flushed_total counter +tempo_live_store_local_blocks_flushed_total 1 +# HELP tempo_live_store_local_failed_flushes_total The total number of failed local complete block flushes +# TYPE tempo_live_store_local_failed_flushes_total counter +tempo_live_store_local_failed_flushes_total 0 +# HELP tempo_live_store_local_flush_duration_seconds Records the amount of time to flush a local complete block. +# TYPE tempo_live_store_local_flush_duration_seconds histogram +tempo_live_store_local_flush_duration_seconds_bucket{le="1"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="2"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="4"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="8"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="16"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="32"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="64"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="128"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="256"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="512"} 1 +tempo_live_store_local_flush_duration_seconds_bucket{le="+Inf"} 1 +tempo_live_store_local_flush_duration_seconds_sum 0.057309096 +tempo_live_store_local_flush_duration_seconds_count 1 +# HELP tempo_live_store_local_flush_failed_retries_total The total number of failed retries after a failed local complete block flush +# TYPE tempo_live_store_local_flush_failed_retries_total counter +tempo_live_store_local_flush_failed_retries_total 0 +# HELP tempo_live_store_local_flush_retries_total The total number of retries after a failed local complete block flush +# TYPE tempo_live_store_local_flush_retries_total counter +tempo_live_store_local_flush_retries_total 0 +# HELP tempo_live_store_local_flush_size_bytes Size in bytes of local complete blocks flushed. +# TYPE tempo_live_store_local_flush_size_bytes histogram +tempo_live_store_local_flush_size_bytes_bucket{le="1.048576e+06"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="2.097152e+06"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="4.194304e+06"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="8.388608e+06"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="1.6777216e+07"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="3.3554432e+07"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="6.7108864e+07"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="1.34217728e+08"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="2.68435456e+08"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="5.36870912e+08"} 1 +tempo_live_store_local_flush_size_bytes_bucket{le="+Inf"} 1 +tempo_live_store_local_flush_size_bytes_sum 26376 +tempo_live_store_local_flush_size_bytes_count 1 +# HELP tempo_live_store_query_inspected_bytes_total Total bytes inspected by live-store queries, per tenant and operation. +# TYPE tempo_live_store_query_inspected_bytes_total counter +tempo_live_store_query_inspected_bytes_total{op="search",tenant="single-tenant"} 15950 +tempo_live_store_query_inspected_bytes_total{op="trace_by_id",tenant="single-tenant"} 192488 +# HELP tempo_live_store_ready 1 if ready to serve queries, 0 otherwise +# TYPE tempo_live_store_ready gauge +tempo_live_store_ready 0 +# HELP tempo_live_store_traces_created_total The total number of traces created per tenant. +# TYPE tempo_live_store_traces_created_total counter +tempo_live_store_traces_created_total{tenant="single-tenant"} 4 +# HELP tempo_metrics_generator_assigned_partitions Number of Kafka partitions currently assigned to this generator instance. +# TYPE tempo_metrics_generator_assigned_partitions gauge +tempo_metrics_generator_assigned_partitions 0 +# HELP tempo_metrics_generator_enqueue_time_seconds_total The total amount of time spent waiting to enqueue for processing +# TYPE tempo_metrics_generator_enqueue_time_seconds_total counter +tempo_metrics_generator_enqueue_time_seconds_total 0 +# HELP tempo_overrides_user_configurable_overrides_list_total How often the user-configurable overrides was listed +# TYPE tempo_overrides_user_configurable_overrides_list_total counter +tempo_overrides_user_configurable_overrides_list_total 0 +# HELP tempo_overrides_user_configurable_overrides_reload_failed_total How often reloading the user-configurable overrides has failed +# TYPE tempo_overrides_user_configurable_overrides_reload_failed_total counter +tempo_overrides_user_configurable_overrides_reload_failed_total 0 +# HELP tempo_partition_ring_lifecycler_reconciles_total Total number of reconciliations started. +# TYPE tempo_partition_ring_lifecycler_reconciles_total counter +tempo_partition_ring_lifecycler_reconciles_total{name="livestore-partitions",type="other-partitions"} 28121 +tempo_partition_ring_lifecycler_reconciles_total{name="livestore-partitions",type="owned-partition"} 28121 +# HELP tempo_partition_ring_partitions Number of partitions by state in the partitions ring. +# TYPE tempo_partition_ring_partitions gauge +tempo_partition_ring_partitions{name="livestore-partitions",state="Active"} 1 +tempo_partition_ring_partitions{name="livestore-partitions",state="Inactive"} 0 +tempo_partition_ring_partitions{name="livestore-partitions",state="Pending"} 0 +# HELP tempo_querier_backend_processing_duration_seconds Time the querier spends processing backend blocks (object-store scan + interleaved I/O), by operation and tenant. Excludes recent (live-store) data. +# TYPE tempo_querier_backend_processing_duration_seconds histogram +tempo_querier_backend_processing_duration_seconds_bucket{operation="traces",tenant="single-tenant",le="0.005"} 16 +tempo_querier_backend_processing_duration_seconds_bucket{operation="traces",tenant="single-tenant",le="0.02"} 16 +tempo_querier_backend_processing_duration_seconds_bucket{operation="traces",tenant="single-tenant",le="0.08"} 16 +tempo_querier_backend_processing_duration_seconds_bucket{operation="traces",tenant="single-tenant",le="0.32"} 16 +tempo_querier_backend_processing_duration_seconds_bucket{operation="traces",tenant="single-tenant",le="1.28"} 16 +tempo_querier_backend_processing_duration_seconds_bucket{operation="traces",tenant="single-tenant",le="5.12"} 16 +tempo_querier_backend_processing_duration_seconds_bucket{operation="traces",tenant="single-tenant",le="+Inf"} 16 +tempo_querier_backend_processing_duration_seconds_sum{operation="traces",tenant="single-tenant"} 0.00010650600000000001 +tempo_querier_backend_processing_duration_seconds_count{operation="traces",tenant="single-tenant"} 16 +# HELP tempo_querier_livestore_clients The current number of livestore clients. +# TYPE tempo_querier_livestore_clients gauge +tempo_querier_livestore_clients 1 +# HELP tempo_querier_worker_request_executed_total The total number of requests executed by the querier worker. +# TYPE tempo_querier_worker_request_executed_total counter +tempo_querier_worker_request_executed_total 35 +# HELP tempo_query_frontend_actual_batch_size Batch size. +# TYPE tempo_query_frontend_actual_batch_size histogram +tempo_query_frontend_actual_batch_size_bucket{le="1"} 21 +tempo_query_frontend_actual_batch_size_bucket{le="2.4"} 28 +tempo_query_frontend_actual_batch_size_bucket{le="3.8"} 28 +tempo_query_frontend_actual_batch_size_bucket{le="5.199999999999999"} 28 +tempo_query_frontend_actual_batch_size_bucket{le="6.6"} 28 +tempo_query_frontend_actual_batch_size_bucket{le="+Inf"} 28 +tempo_query_frontend_actual_batch_size_sum 35 +tempo_query_frontend_actual_batch_size_count 28 +# HELP tempo_query_frontend_batch_weight Weight of the batch. +# TYPE tempo_query_frontend_batch_weight histogram +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="1"} 1 +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="2"} 22 +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="3"} 22 +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="4"} 28 +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="5"} 28 +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="6"} 28 +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="7"} 28 +tempo_query_frontend_batch_weight_bucket{user="single-tenant",le="+Inf"} 28 +tempo_query_frontend_batch_weight_sum{user="single-tenant"} 67 +tempo_query_frontend_batch_weight_count{user="single-tenant"} 28 +# HELP tempo_query_frontend_bytes_inspected_total Bytes read from storage using queries per tenant +# TYPE tempo_query_frontend_bytes_inspected_total counter +tempo_query_frontend_bytes_inspected_total{op="search",tenant="single-tenant"} 15950 +tempo_query_frontend_bytes_inspected_total{op="traces",tenant="single-tenant"} 192488 +# HELP tempo_query_frontend_connected_clients Number of worker clients currently connected to the frontend. +# TYPE tempo_query_frontend_connected_clients gauge +tempo_query_frontend_connected_clients 0 +# HELP tempo_query_frontend_jobs_per_query Number of planned jobs per query in the query frontend. +# TYPE tempo_query_frontend_jobs_per_query histogram +tempo_query_frontend_jobs_per_query_bucket{op="search",le="1"} 0 +tempo_query_frontend_jobs_per_query_bucket{op="search",le="10"} 1 +tempo_query_frontend_jobs_per_query_bucket{op="search",le="100"} 1 +tempo_query_frontend_jobs_per_query_bucket{op="search",le="1000"} 1 +tempo_query_frontend_jobs_per_query_bucket{op="search",le="10000"} 1 +tempo_query_frontend_jobs_per_query_bucket{op="search",le="100000"} 1 +tempo_query_frontend_jobs_per_query_bucket{op="search",le="1e+06"} 1 +tempo_query_frontend_jobs_per_query_bucket{op="search",le="+Inf"} 1 +tempo_query_frontend_jobs_per_query_sum{op="search"} 3 +tempo_query_frontend_jobs_per_query_count{op="search"} 1 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="1"} 0 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="10"} 16 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="100"} 16 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="1000"} 16 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="10000"} 16 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="100000"} 16 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="1e+06"} 16 +tempo_query_frontend_jobs_per_query_bucket{op="traces",le="+Inf"} 16 +tempo_query_frontend_jobs_per_query_sum{op="traces"} 32 +tempo_query_frontend_jobs_per_query_count{op="traces"} 16 +# HELP tempo_query_frontend_queries_total Total queries received per tenant. +# TYPE tempo_query_frontend_queries_total counter +tempo_query_frontend_queries_total{op="search",result="completed",tenant="single-tenant"} 1 +tempo_query_frontend_queries_total{op="traces",result="completed",tenant="single-tenant"} 16 +# HELP tempo_query_frontend_queries_within_slo_total Total Queries within SLO per tenant +# TYPE tempo_query_frontend_queries_within_slo_total counter +tempo_query_frontend_queries_within_slo_total{op="search",result="completed",tenant="single-tenant"} 1 +tempo_query_frontend_queries_within_slo_total{op="traces",result="completed",tenant="single-tenant"} 16 +# HELP tempo_query_frontend_queue_duration_seconds Time spend by requests queued. +# TYPE tempo_query_frontend_queue_duration_seconds histogram +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="0.005"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="0.01"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="0.025"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="0.05"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="0.1"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="0.25"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="0.5"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="1"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="2.5"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="5"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="10"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="search",le="+Inf"} 3 +tempo_query_frontend_queue_duration_seconds_sum{op="search"} 8.681099999999999e-05 +tempo_query_frontend_queue_duration_seconds_count{op="search"} 3 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="0.005"} 31 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="0.01"} 31 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="0.025"} 31 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="0.05"} 31 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="0.1"} 31 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="0.25"} 32 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="0.5"} 32 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="1"} 32 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="2.5"} 32 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="5"} 32 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="10"} 32 +tempo_query_frontend_queue_duration_seconds_bucket{op="traces",le="+Inf"} 32 +tempo_query_frontend_queue_duration_seconds_sum{op="traces"} 0.10746361399999999 +tempo_query_frontend_queue_duration_seconds_count{op="traces"} 32 +# HELP tempo_query_frontend_queue_length Number of queries in the queue. +# TYPE tempo_query_frontend_queue_length gauge +tempo_query_frontend_queue_length{user="single-tenant"} 0 +# HELP tempo_query_frontend_retries Number of times a request is retried. +# TYPE tempo_query_frontend_retries histogram +tempo_query_frontend_retries_bucket{le="0"} 35 +tempo_query_frontend_retries_bucket{le="1"} 35 +tempo_query_frontend_retries_bucket{le="2"} 35 +tempo_query_frontend_retries_bucket{le="3"} 35 +tempo_query_frontend_retries_bucket{le="4"} 35 +tempo_query_frontend_retries_bucket{le="5"} 35 +tempo_query_frontend_retries_bucket{le="+Inf"} 35 +tempo_query_frontend_retries_sum 0 +tempo_query_frontend_retries_count 35 +# HELP tempo_receiver_accepted_spans Number of spans successfully pushed into the pipeline. +# TYPE tempo_receiver_accepted_spans counter +tempo_receiver_accepted_spans{receiver="otlp/otlp_receiver",transport="http"} 4 +# HELP tempo_receiver_refused_spans Number of spans that could not be pushed into the pipeline. +# TYPE tempo_receiver_refused_spans counter +tempo_receiver_refused_spans{receiver="otlp/otlp_receiver",transport="http"} 0 +# HELP tempo_request_duration_seconds Time (in seconds) spent serving HTTP requests. +# TYPE tempo_request_duration_seconds histogram +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="0.01"} 0 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="0.025"} 0 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="0.05"} 0 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="0.1"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="0.25"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="0.5"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="1"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="1.5"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="2.5"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="5"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="10"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="25"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="50"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="100"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_search",status_code="200",ws="false",le="+Inf"} 1 +tempo_request_duration_seconds_sum{method="GET",route="api_search",status_code="200",ws="false"} 0.072098933 +tempo_request_duration_seconds_count{method="GET",route="api_search",status_code="200",ws="false"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="0.01"} 14 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="0.025"} 14 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="0.05"} 15 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="0.1"} 15 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="0.25"} 15 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="0.5"} 15 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="1"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="1.5"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="2.5"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="5"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="10"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="25"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="50"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="100"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false",le="+Inf"} 16 +tempo_request_duration_seconds_sum{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false"} 0.7184611129999999 +tempo_request_duration_seconds_count{method="GET",route="api_v2_traces_traceid",status_code="200",ws="false"} 16 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="0.01"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="0.025"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="0.05"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="0.1"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="0.25"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="0.5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="1"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="1.5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="2.5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="10"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="25"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="50"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="100"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="metrics",status_code="200",ws="false",le="+Inf"} 2 +tempo_request_duration_seconds_sum{method="GET",route="metrics",status_code="200",ws="false"} 0.013940889 +tempo_request_duration_seconds_count{method="GET",route="metrics",status_code="200",ws="false"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="0.01"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="0.025"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="0.05"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="0.1"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="0.25"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="0.5"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="1"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="1.5"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="2.5"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="5"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="10"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="25"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="50"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="100"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_search",status_code="200",ws="false",le="+Inf"} 3 +tempo_request_duration_seconds_sum{method="GET",route="querier_api_search",status_code="200",ws="false"} 0.005661401 +tempo_request_duration_seconds_count{method="GET",route="querier_api_search",status_code="200",ws="false"} 3 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="0.01"} 29 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="0.025"} 30 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="0.05"} 31 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="0.1"} 31 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="0.25"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="0.5"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="1"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="1.5"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="2.5"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="5"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="10"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="25"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="50"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="100"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false",le="+Inf"} 32 +tempo_request_duration_seconds_sum{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false"} 0.27683221900000005 +tempo_request_duration_seconds_count{method="GET",route="querier_api_v2_traces_traceid",status_code="200",ws="false"} 32 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="0.01"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="0.025"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="0.05"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="0.1"} 1 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="0.25"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="0.5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="1"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="1.5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="2.5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="5"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="10"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="25"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="50"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="100"} 2 +tempo_request_duration_seconds_bucket{method="GET",route="ready",status_code="200",ws="false",le="+Inf"} 2 +tempo_request_duration_seconds_sum{method="GET",route="ready",status_code="200",ws="false"} 0.107151896 +tempo_request_duration_seconds_count{method="GET",route="ready",status_code="200",ws="false"} 2 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="0.01"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="0.025"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="0.05"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="0.1"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="0.25"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="0.5"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="1"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="1.5"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="2.5"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="5"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="10"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="25"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="50"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="100"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false",le="+Inf"} 1 +tempo_request_duration_seconds_sum{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false"} 3.9204e-05 +tempo_request_duration_seconds_count{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",status_code="success",ws="false"} 1 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="0.01"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="0.025"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="0.05"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="0.1"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="0.25"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="0.5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="1"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="1.5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="2.5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="10"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="25"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="50"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="100"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false",le="+Inf"} 20 +tempo_request_duration_seconds_sum{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false"} 2.8120932362759113e+06 +tempo_request_duration_seconds_count{method="gRPC",route="/frontend.Frontend/Process",status_code="error",ws="false"} 20 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="0.01"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="0.025"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="0.05"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="0.1"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="0.25"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="0.5"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="1"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="1.5"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="2.5"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="5"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="10"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="25"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="50"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="100"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false",le="+Inf"} 13 +tempo_request_duration_seconds_sum{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false"} 0.000252134 +tempo_request_duration_seconds_count{method="gRPC",route="/grpc.health.v1.Health/Check",status_code="success",ws="false"} 13 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="0.01"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="0.025"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="0.05"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="0.1"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="0.25"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="0.5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="1"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="1.5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="2.5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="5"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="10"} 0 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="25"} 1984 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="50"} 1984 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="100"} 1984 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false",le="+Inf"} 1984 +tempo_request_duration_seconds_sum{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false"} 29760.209185702006 +tempo_request_duration_seconds_count{method="gRPC",route="/tempopb.BackendScheduler/Next",status_code="error",ws="false"} 1984 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="0.01"} 15 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="0.025"} 15 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="0.05"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="0.1"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="0.25"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="0.5"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="1"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="1.5"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="2.5"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="5"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="10"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="25"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="50"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="100"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false",le="+Inf"} 16 +tempo_request_duration_seconds_sum{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false"} 0.082101172 +tempo_request_duration_seconds_count{method="gRPC",route="/tempopb.Querier/FindTraceByID",status_code="success",ws="false"} 16 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="0.01"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="0.025"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="0.05"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="0.1"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="0.25"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="0.5"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="1"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="1.5"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="2.5"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="5"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="10"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="25"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="50"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="100"} 3 +tempo_request_duration_seconds_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false",le="+Inf"} 3 +tempo_request_duration_seconds_sum{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false"} 0.003434826 +tempo_request_duration_seconds_count{method="gRPC",route="/tempopb.Querier/SearchRecent",status_code="success",ws="false"} 3 +# HELP tempo_request_message_bytes Size (in bytes) of messages received in the request. +# TYPE tempo_request_message_bytes histogram +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="4"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="16"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="64"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="256"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="1024"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="4096"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="16384"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="65536"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="262144"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="1.048576e+06"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="4.194304e+06"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="1.6777216e+07"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="6.7108864e+07"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="2.68435456e+08"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="1.073741824e+09"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_search",le="+Inf"} 1 +tempo_request_message_bytes_sum{method="GET",route="api_search"} 0 +tempo_request_message_bytes_count{method="GET",route="api_search"} 1 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="4"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="16"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="64"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="256"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1024"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="4096"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="16384"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="65536"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="262144"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1.048576e+06"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="4.194304e+06"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1.6777216e+07"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="6.7108864e+07"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="2.68435456e+08"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1.073741824e+09"} 16 +tempo_request_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="+Inf"} 16 +tempo_request_message_bytes_sum{method="GET",route="api_v2_traces_traceid"} 0 +tempo_request_message_bytes_count{method="GET",route="api_v2_traces_traceid"} 16 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="4"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="16"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="64"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="256"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="1024"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="4096"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="16384"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="65536"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="262144"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="1.048576e+06"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="4.194304e+06"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="1.6777216e+07"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="6.7108864e+07"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="2.68435456e+08"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="1.073741824e+09"} 2 +tempo_request_message_bytes_bucket{method="GET",route="metrics",le="+Inf"} 2 +tempo_request_message_bytes_sum{method="GET",route="metrics"} 0 +tempo_request_message_bytes_count{method="GET",route="metrics"} 2 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="4"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="16"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="64"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="256"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="1024"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="4096"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="16384"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="65536"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="262144"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="1.048576e+06"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="4.194304e+06"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="1.6777216e+07"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="6.7108864e+07"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="2.68435456e+08"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="1.073741824e+09"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_search",le="+Inf"} 3 +tempo_request_message_bytes_sum{method="GET",route="querier_api_search"} 0 +tempo_request_message_bytes_count{method="GET",route="querier_api_search"} 3 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="4"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="16"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="64"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="256"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1024"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="4096"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="16384"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="65536"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="262144"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1.048576e+06"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="4.194304e+06"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1.6777216e+07"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="6.7108864e+07"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="2.68435456e+08"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1.073741824e+09"} 32 +tempo_request_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="+Inf"} 32 +tempo_request_message_bytes_sum{method="GET",route="querier_api_v2_traces_traceid"} 0 +tempo_request_message_bytes_count{method="GET",route="querier_api_v2_traces_traceid"} 32 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="4"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="16"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="64"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="256"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="1024"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="4096"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="16384"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="65536"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="262144"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="1.048576e+06"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="4.194304e+06"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="1.6777216e+07"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="6.7108864e+07"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="2.68435456e+08"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="1.073741824e+09"} 2 +tempo_request_message_bytes_bucket{method="GET",route="ready",le="+Inf"} 2 +tempo_request_message_bytes_sum{method="GET",route="ready"} 0 +tempo_request_message_bytes_count{method="GET",route="ready"} 2 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="4"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="16"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="64"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="256"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1024"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="4096"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="16384"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="65536"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="262144"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1.048576e+06"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="4.194304e+06"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1.6777216e+07"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="6.7108864e+07"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="2.68435456e+08"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1.073741824e+09"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="+Inf"} 1 +tempo_request_message_bytes_sum{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown"} 37 +tempo_request_message_bytes_count{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown"} 1 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="4"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="16"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="64"} 20 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="256"} 39 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1024"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="4096"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="16384"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="65536"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="262144"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1.048576e+06"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="4.194304e+06"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1.6777216e+07"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="6.7108864e+07"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="2.68435456e+08"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1.073741824e+09"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="+Inf"} 48 +tempo_request_message_bytes_sum{method="gRPC",route="/frontend.Frontend/Process"} 6520 +tempo_request_message_bytes_count{method="gRPC",route="/frontend.Frontend/Process"} 48 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="4"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="16"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="64"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="256"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1024"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="4096"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="16384"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="65536"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="262144"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1.048576e+06"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="4.194304e+06"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1.6777216e+07"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="6.7108864e+07"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="2.68435456e+08"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1.073741824e+09"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="+Inf"} 13 +tempo_request_message_bytes_sum{method="gRPC",route="/grpc.health.v1.Health/Check"} 65 +tempo_request_message_bytes_count{method="gRPC",route="/grpc.health.v1.Health/Check"} 13 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="4"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="16"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="64"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="256"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1024"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="4096"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="16384"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="65536"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="262144"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1.048576e+06"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="4.194304e+06"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1.6777216e+07"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="6.7108864e+07"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="2.68435456e+08"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1.073741824e+09"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="+Inf"} 1984 +tempo_request_message_bytes_sum{method="gRPC",route="/tempopb.BackendScheduler/Next"} 73408 +tempo_request_message_bytes_count{method="gRPC",route="/tempopb.BackendScheduler/Next"} 1984 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="4"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="16"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="64"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="256"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1024"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="4096"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="16384"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="65536"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="262144"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1.048576e+06"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="4.194304e+06"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1.6777216e+07"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="6.7108864e+07"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="2.68435456e+08"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1.073741824e+09"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="+Inf"} 16 +tempo_request_message_bytes_sum{method="gRPC",route="/tempopb.Querier/FindTraceByID"} 1072 +tempo_request_message_bytes_count{method="gRPC",route="/tempopb.Querier/FindTraceByID"} 16 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="4"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="16"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="64"} 0 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="256"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1024"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="4096"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="16384"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="65536"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="262144"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1.048576e+06"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="4.194304e+06"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1.6777216e+07"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="6.7108864e+07"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="2.68435456e+08"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1.073741824e+09"} 3 +tempo_request_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="+Inf"} 3 +tempo_request_message_bytes_sum{method="gRPC",route="/tempopb.Querier/SearchRecent"} 243 +tempo_request_message_bytes_count{method="gRPC",route="/tempopb.Querier/SearchRecent"} 3 +# HELP tempo_response_message_bytes Size (in bytes) of messages sent in response. +# TYPE tempo_response_message_bytes histogram +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="4"} 0 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="16"} 0 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="64"} 0 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="256"} 0 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="1024"} 0 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="4096"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="16384"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="65536"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="262144"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="1.048576e+06"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="4.194304e+06"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="1.6777216e+07"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="6.7108864e+07"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="2.68435456e+08"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="1.073741824e+09"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_search",le="+Inf"} 1 +tempo_response_message_bytes_sum{method="GET",route="api_search"} 2517 +tempo_response_message_bytes_count{method="GET",route="api_search"} 1 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="4"} 0 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="16"} 0 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="64"} 8 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="256"} 8 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1024"} 12 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="4096"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="16384"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="65536"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="262144"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1.048576e+06"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="4.194304e+06"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1.6777216e+07"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="6.7108864e+07"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="2.68435456e+08"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="1.073741824e+09"} 16 +tempo_response_message_bytes_bucket{method="GET",route="api_v2_traces_traceid",le="+Inf"} 16 +tempo_response_message_bytes_sum{method="GET",route="api_v2_traces_traceid"} 8062 +tempo_response_message_bytes_count{method="GET",route="api_v2_traces_traceid"} 16 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="4"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="16"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="64"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="256"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="1024"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="4096"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="16384"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="65536"} 0 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="262144"} 2 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="1.048576e+06"} 2 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="4.194304e+06"} 2 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="1.6777216e+07"} 2 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="6.7108864e+07"} 2 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="2.68435456e+08"} 2 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="1.073741824e+09"} 2 +tempo_response_message_bytes_bucket{method="GET",route="metrics",le="+Inf"} 2 +tempo_response_message_bytes_sum{method="GET",route="metrics"} 191771 +tempo_response_message_bytes_count{method="GET",route="metrics"} 2 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="4"} 2 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="16"} 2 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="64"} 2 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="256"} 2 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="1024"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="4096"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="16384"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="65536"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="262144"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="1.048576e+06"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="4.194304e+06"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="1.6777216e+07"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="6.7108864e+07"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="2.68435456e+08"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="1.073741824e+09"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_search",le="+Inf"} 3 +tempo_response_message_bytes_sum{method="GET",route="querier_api_search"} 921 +tempo_response_message_bytes_count{method="GET",route="querier_api_search"} 3 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="4"} 24 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="16"} 24 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="64"} 24 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="256"} 26 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1024"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="4096"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="16384"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="65536"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="262144"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1.048576e+06"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="4.194304e+06"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1.6777216e+07"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="6.7108864e+07"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="2.68435456e+08"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="1.073741824e+09"} 32 +tempo_response_message_bytes_bucket{method="GET",route="querier_api_v2_traces_traceid",le="+Inf"} 32 +tempo_response_message_bytes_sum{method="GET",route="querier_api_v2_traces_traceid"} 3486 +tempo_response_message_bytes_count{method="GET",route="querier_api_v2_traces_traceid"} 32 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="4"} 0 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="16"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="64"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="256"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="1024"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="4096"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="16384"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="65536"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="262144"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="1.048576e+06"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="4.194304e+06"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="1.6777216e+07"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="6.7108864e+07"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="2.68435456e+08"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="1.073741824e+09"} 2 +tempo_response_message_bytes_bucket{method="GET",route="ready",le="+Inf"} 2 +tempo_response_message_bytes_sum{method="GET",route="ready"} 12 +tempo_response_message_bytes_count{method="GET",route="ready"} 2 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="4"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="16"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="64"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="256"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1024"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="4096"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="16384"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="65536"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="262144"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1.048576e+06"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="4.194304e+06"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1.6777216e+07"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="6.7108864e+07"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="2.68435456e+08"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="1.073741824e+09"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown",le="+Inf"} 1 +tempo_response_message_bytes_sum{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown"} 5 +tempo_response_message_bytes_count{method="gRPC",route="/frontend.Frontend/NotifyClientShutdown"} 1 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="4"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="16"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="64"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="256"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1024"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="4096"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="16384"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="65536"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="262144"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1.048576e+06"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="4.194304e+06"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1.6777216e+07"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="6.7108864e+07"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="2.68435456e+08"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="1.073741824e+09"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/frontend.Frontend/Process",le="+Inf"} 48 +tempo_response_message_bytes_sum{method="gRPC",route="/frontend.Frontend/Process"} 6862 +tempo_response_message_bytes_count{method="gRPC",route="/frontend.Frontend/Process"} 48 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="4"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="16"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="64"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="256"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1024"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="4096"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="16384"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="65536"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="262144"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1.048576e+06"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="4.194304e+06"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1.6777216e+07"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="6.7108864e+07"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="2.68435456e+08"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="1.073741824e+09"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/grpc.health.v1.Health/Check",le="+Inf"} 13 +tempo_response_message_bytes_sum{method="gRPC",route="/grpc.health.v1.Health/Check"} 325 +tempo_response_message_bytes_count{method="gRPC",route="/grpc.health.v1.Health/Check"} 13 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="4"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="16"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="64"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="256"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1024"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="4096"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="16384"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="65536"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="262144"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1.048576e+06"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="4.194304e+06"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1.6777216e+07"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="6.7108864e+07"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="2.68435456e+08"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="1.073741824e+09"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.BackendScheduler/Next",le="+Inf"} 0 +tempo_response_message_bytes_sum{method="gRPC",route="/tempopb.BackendScheduler/Next"} 0 +tempo_response_message_bytes_count{method="gRPC",route="/tempopb.BackendScheduler/Next"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="4"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="16"} 8 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="64"} 8 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="256"} 8 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1024"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="4096"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="16384"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="65536"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="262144"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1.048576e+06"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="4.194304e+06"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1.6777216e+07"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="6.7108864e+07"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="2.68435456e+08"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="1.073741824e+09"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/FindTraceByID",le="+Inf"} 16 +tempo_response_message_bytes_sum{method="gRPC",route="/tempopb.Querier/FindTraceByID"} 3358 +tempo_response_message_bytes_count{method="gRPC",route="/tempopb.Querier/FindTraceByID"} 16 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="4"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="16"} 0 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="64"} 2 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="256"} 2 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1024"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="4096"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="16384"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="65536"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="262144"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1.048576e+06"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="4.194304e+06"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1.6777216e+07"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="6.7108864e+07"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="2.68435456e+08"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="1.073741824e+09"} 3 +tempo_response_message_bytes_bucket{method="gRPC",route="/tempopb.Querier/SearchRecent",le="+Inf"} 3 +tempo_response_message_bytes_sum{method="gRPC",route="/tempopb.Querier/SearchRecent"} 472 +tempo_response_message_bytes_count{method="gRPC",route="/tempopb.Querier/SearchRecent"} 3 +# HELP tempo_ring_member_heartbeats_total The total number of heartbeats sent. +# TYPE tempo_ring_member_heartbeats_total counter +tempo_ring_member_heartbeats_total{name="live-store"} 28120 +# HELP tempo_ring_member_tokens_owned The number of tokens owned in the ring. +# TYPE tempo_ring_member_tokens_owned gauge +tempo_ring_member_tokens_owned{name="live-store"} 0 +# HELP tempo_ring_member_tokens_to_own The number of tokens to own in the ring. +# TYPE tempo_ring_member_tokens_to_own gauge +tempo_ring_member_tokens_to_own{name="live-store"} 0 +# HELP tempo_ring_members Number of members in the ring +# TYPE tempo_ring_members gauge +tempo_ring_members{name="live-store",state="ACTIVE"} 0 +tempo_ring_members{name="live-store",state="JOINING"} 0 +tempo_ring_members{name="live-store",state="LEAVING"} 0 +tempo_ring_members{name="live-store",state="PENDING"} 0 +tempo_ring_members{name="live-store",state="Unhealthy"} 0 +# HELP tempo_ring_oldest_member_timestamp Timestamp of the oldest member in the ring. +# TYPE tempo_ring_oldest_member_timestamp gauge +tempo_ring_oldest_member_timestamp{name="live-store",state="ACTIVE"} 0 +tempo_ring_oldest_member_timestamp{name="live-store",state="JOINING"} 0 +tempo_ring_oldest_member_timestamp{name="live-store",state="LEAVING"} 0 +tempo_ring_oldest_member_timestamp{name="live-store",state="PENDING"} 0 +tempo_ring_oldest_member_timestamp{name="live-store",state="Unhealthy"} 0 +# HELP tempo_ring_tokens_total Number of tokens in the ring +# TYPE tempo_ring_tokens_total gauge +tempo_ring_tokens_total{name="live-store"} 0 +# HELP tempo_spans_distance_in_past_seconds The number of seconds in the past of the span end time in relation to the ingestion time. +# TYPE tempo_spans_distance_in_past_seconds histogram +tempo_spans_distance_in_past_seconds_bucket{tenant="single-tenant",le="300"} 4 +tempo_spans_distance_in_past_seconds_bucket{tenant="single-tenant",le="1800"} 4 +tempo_spans_distance_in_past_seconds_bucket{tenant="single-tenant",le="3600"} 4 +tempo_spans_distance_in_past_seconds_bucket{tenant="single-tenant",le="+Inf"} 4 +tempo_spans_distance_in_past_seconds_sum{tenant="single-tenant"} 12 +tempo_spans_distance_in_past_seconds_count{tenant="single-tenant"} 4 +# HELP tempo_tcp_connections Current number of accepted TCP connections. +# TYPE tempo_tcp_connections gauge +tempo_tcp_connections{protocol="grpc"} 2 +tempo_tcp_connections{protocol="http"} 1 +# HELP tempo_tcp_connections_limit The max number of TCP connections that can be accepted (0 means no limit). +# TYPE tempo_tcp_connections_limit gauge +tempo_tcp_connections_limit{protocol="grpc"} 0 +tempo_tcp_connections_limit{protocol="http"} 0 +# HELP tempodb_backend_hedged_roundtrips_total Total number of hedged backend requests. +# TYPE tempodb_backend_hedged_roundtrips_total counter +tempodb_backend_hedged_roundtrips_total 0 +# HELP tempodb_blocklist_poll_duration_seconds Records the amount of time to poll and update the blocklist. +# TYPE tempodb_blocklist_poll_duration_seconds histogram +tempodb_blocklist_poll_duration_seconds_bucket{le="0.005"} 464 +tempodb_blocklist_poll_duration_seconds_bucket{le="0.01"} 464 +tempodb_blocklist_poll_duration_seconds_bucket{le="0.025"} 464 +tempodb_blocklist_poll_duration_seconds_bucket{le="0.05"} 464 +tempodb_blocklist_poll_duration_seconds_bucket{le="0.1"} 465 +tempodb_blocklist_poll_duration_seconds_bucket{le="0.25"} 468 +tempodb_blocklist_poll_duration_seconds_bucket{le="0.5"} 469 +tempodb_blocklist_poll_duration_seconds_bucket{le="1"} 469 +tempodb_blocklist_poll_duration_seconds_bucket{le="2.5"} 469 +tempodb_blocklist_poll_duration_seconds_bucket{le="5"} 469 +tempodb_blocklist_poll_duration_seconds_bucket{le="10"} 469 +tempodb_blocklist_poll_duration_seconds_bucket{le="+Inf"} 469 +tempodb_blocklist_poll_duration_seconds_sum 0.8501230490000001 +tempodb_blocklist_poll_duration_seconds_count 469 +# HELP tempodb_compaction_errors_total Total number of errors occurring during compaction. +# TYPE tempodb_compaction_errors_total counter +tempodb_compaction_errors_total 0 +# HELP tempodb_compaction_output_block_size_bytes Size in bytes of blocks produced by compaction. +# TYPE tempodb_compaction_output_block_size_bytes histogram +tempodb_compaction_output_block_size_bytes_bucket{le="1.048576e+06"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="2.097152e+06"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="4.194304e+06"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="8.388608e+06"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="1.6777216e+07"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="3.3554432e+07"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="6.7108864e+07"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="1.34217728e+08"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="2.68435456e+08"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="5.36870912e+08"} 0 +tempodb_compaction_output_block_size_bytes_bucket{le="+Inf"} 0 +tempodb_compaction_output_block_size_bytes_sum 0 +tempodb_compaction_output_block_size_bytes_count 0 +# HELP tempodb_retention_deleted_total Total number of blocks deleted. +# TYPE tempodb_retention_deleted_total counter +tempodb_retention_deleted_total 0 +# HELP tempodb_retention_duration_seconds Records the amount of time to perform retention tasks. +# TYPE tempodb_retention_duration_seconds histogram +tempodb_retention_duration_seconds_bucket{le="0.25"} 0 +tempodb_retention_duration_seconds_bucket{le="0.5"} 0 +tempodb_retention_duration_seconds_bucket{le="1"} 0 +tempodb_retention_duration_seconds_bucket{le="2"} 0 +tempodb_retention_duration_seconds_bucket{le="4"} 0 +tempodb_retention_duration_seconds_bucket{le="8"} 0 +tempodb_retention_duration_seconds_bucket{le="+Inf"} 0 +tempodb_retention_duration_seconds_sum 0 +tempodb_retention_duration_seconds_count 0 +# HELP tempodb_retention_errors_total Total number of times an error occurred while performing retention tasks. +# TYPE tempodb_retention_errors_total counter +tempodb_retention_errors_total 0 +# HELP tempodb_retention_marked_for_deletion_total Total number of blocks marked for deletion. +# TYPE tempodb_retention_marked_for_deletion_total counter +tempodb_retention_marked_for_deletion_total 0 +# HELP tempodb_work_queue_length Current length of the work queue. +# TYPE tempodb_work_queue_length gauge +tempodb_work_queue_length 0 +# HELP tempodb_work_queue_max Maximum number of items in the work queue. +# TYPE tempodb_work_queue_max gauge +tempodb_work_queue_max 20000 diff --git a/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-after.json b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-after.json new file mode 100644 index 0000000..b2f5879 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-after.json @@ -0,0 +1,205 @@ +{ + "traces": [ + { + "traceID": "99cf4f054710442097ec1b830809ed22", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "submission", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "b4842e47da3649aa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "b4842e47da3649aa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + }, + { + "traceID": "74046adf072c4fb98f00ff8a45206f9e", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "absent", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "8a31626387484695", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "8a31626387484695", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + }, + { + "traceID": "8f5d0e572ac7400f8b9a74a64221564d", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "explicit", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "f0abc9e591da4dfa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "f0abc9e591da4dfa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + }, + { + "traceID": "870d82356c67456b971a98d82395edde", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "zero", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "f8f495dd55794f75", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "f8f495dd55794f75", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + } + ], + "metrics": { + "inspectedBytes": "15950", + "completedJobs": 3, + "totalJobs": 3 + } +} diff --git a/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-before.json b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-before.json new file mode 100644 index 0000000..2af8dde --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-search-before.json @@ -0,0 +1,205 @@ +{ + "traces": [ + { + "traceID": "870d82356c67456b971a98d82395edde", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "zero", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "f8f495dd55794f75", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "f8f495dd55794f75", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + }, + { + "traceID": "99cf4f054710442097ec1b830809ed22", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "submission", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "b4842e47da3649aa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "b4842e47da3649aa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + }, + { + "traceID": "8f5d0e572ac7400f8b9a74a64221564d", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "explicit", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "f0abc9e591da4dfa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "f0abc9e591da4dfa", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + }, + { + "traceID": "74046adf072c4fb98f00ff8a45206f9e", + "rootServiceName": "inkcre-projection-probe", + "rootTraceName": "absent", + "startTimeUnixNano": "1791004150502892000", + "spanSet": { + "spans": [ + { + "spanID": "8a31626387484695", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + }, + "spanSets": [ + { + "spans": [ + { + "spanID": "8a31626387484695", + "startTimeUnixNano": "1791004150502892000", + "durationNanos": "1000", + "attributes": [ + { + "key": "inkcre.job.id", + "value": { + "intValue": "42" + } + } + ] + } + ], + "matched": 1 + } + ], + "serviceStats": { + "inkcre-projection-probe": { + "spanCount": 1 + } + } + } + ], + "metrics": { + "inspectedBytes": "15950", + "completedJobs": 3, + "totalJobs": 3 + } +} diff --git a/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-stats.jsonl b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-stats.jsonl new file mode 100644 index 0000000..6d6928f --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/tempo-20261003/tempo-stats.jsonl @@ -0,0 +1,2 @@ +{"BlockIO":"6.13MB / 8.19kB","CPUPerc":"0.24%","Container":"inkcre-o11y-tempo-g1-b0a97f7c-tempo-1","ID":"b200b42280e8","MemPerc":"1.13%","MemUsage":"40.66MiB / 3.5GiB","Name":"inkcre-o11y-tempo-g1-b0a97f7c-tempo-1","NetIO":"4.14kB / 9.55kB","PIDs":"8"} +{"BlockIO":"272MB / 1MB","CPUPerc":"0.04%","Container":"inkcre-o11y-tempo-g1-b0a97f7c-collector-1","ID":"8edc681c5c01","MemPerc":"5.04%","MemUsage":"25.82MiB / 512MiB","Name":"inkcre-o11y-tempo-g1-b0a97f7c-collector-1","NetIO":"6.8kB / 3.01kB","PIDs":"8"} diff --git a/tasks/observability-foundation/experiments/failure.py b/tasks/observability-foundation/experiments/failure.py new file mode 100644 index 0000000..5e0333c --- /dev/null +++ b/tasks/observability-foundation/experiments/failure.py @@ -0,0 +1,115 @@ +"""Bounded backend outage probe. This does not run an InKCre business operation.""" + +from concurrent.futures import ThreadPoolExecutor +import json +from pathlib import Path +import time +import urllib.request +import uuid + +from lab import docker, PROJECT + +HERE = Path(__file__).resolve().parent + + +def metrics(): + with urllib.request.urlopen("http://127.0.0.1:38888/metrics", timeout=5) as response: + return response.read().decode() + + +def send(index): + now = time.time_ns() + spans = [ + { + "name": "fault.probe", + "traceId": uuid.uuid4().hex, + "spanId": uuid.uuid4().hex[:16], + "startTimeUnixNano": str(now), + "endTimeUnixNano": str(now + 1000000), + "attributes": [ + {"key": "inkcre.lab.padding", "value": {"stringValue": "synthetic-" * 228}} + ], + } + for _ in range(256) + ] + data = { + "resourceSpans": [ + { + "resource": { + "attributes": [ + {"key": "service.name", "value": {"stringValue": "inkcre-o11y-outage"}} + ] + }, + "scopeSpans": [{"spans": spans}], + } + ] + } + request = urllib.request.Request( + "http://127.0.0.1:34318/v1/traces", + data=json.dumps(data).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=10) as response: + return {"request": index, "status": response.status, "body": response.read().decode()} + + +def main(): + before = metrics() + responses = [] + samples = [] + docker("stop", PROJECT + "-openobserve-1") + try: + with ThreadPoolExecutor(max_workers=4) as pool: + responses = list(pool.map(send, range(16))) + for _ in range(15): + samples.append(metrics()) + time.sleep(1) + stats = docker( + "stats", "--no-stream", "--format", "{{json .}}", PROJECT + "-collector-1" + ) + finally: + docker("start", PROJECT + "-openobserve-1") + after = metrics() + evidence = { + "submitted_spans": 4096, + "before": before, + "samples": samples, + "after": after, + "responses": responses, + "collector_stats_after_burst": stats, + } + (HERE / "runtime/failure.json").write_text(json.dumps(evidence, indent=2) + "\n") + interesting = [ + line + for line in after.splitlines() + if not line.startswith("#") + and any( + key in line + for key in [ + "enqueue_failed_spans", + "send_failed_spans", + "queue_size", + "queue_capacity", + "process_memory_rss", + ] + ) + ] + assert any( + "enqueue_failed_spans" in line and float(line.rsplit(" ", 1)[1]) > 0 + for line in interesting + ), "Outage did not exercise queue overflow" + print( + json.dumps( + { + "http_200_responses": sum(r["status"] == 200 for r in responses), + "submitted_spans": 4096, + "observed": interesting, + "business_failure_isolation": "not exercised", + }, + indent=2, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/tasks/observability-foundation/experiments/foundation-probe.py b/tasks/observability-foundation/experiments/foundation-probe.py new file mode 100644 index 0000000..eeb7e7d --- /dev/null +++ b/tasks/observability-foundation/experiments/foundation-probe.py @@ -0,0 +1,518 @@ +"""Real OTLP boundary probe plus temporary SDK fault injection; no external credentials.""" + +import asyncio +from contextlib import ExitStack, redirect_stderr +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import io +import json +import logging +import os +from pathlib import Path +import sys +import threading +from tempfile import TemporaryDirectory +import time +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) +os.environ["INKCRE_ENV_FILE"] = "" +os.environ["OTEL_EXPORTER_OTLP_TIMEOUT"] = "1" +for _signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{_signal}_TIMEOUT"] = "1" +os.environ["OTEL_PYTHON_SDK_INTERNAL_METRICS_ENABLED"] = "true" + +from google.protobuf.json_format import MessageToDict +from opentelemetry.proto.collector.logs.v1.logs_service_pb2 import ExportLogsServiceRequest +from opentelemetry.proto.collector.metrics.v1.metrics_service_pb2 import ( + ExportMetricsServiceRequest, +) +from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( + ExportTraceServiceRequest, +) +from opentelemetry.trace import ( + NonRecordingSpan, + SpanContext, + TraceState, + get_current_span, + use_span, +) +from libs.obsrv import telemetry + +CANARY = "foundation-private-content-should-not-export" +records = [] +delay = 0.0 + + +class Receiver(BaseHTTPRequestHandler): + def do_POST(self): + body = self.rfile.read(int(self.headers["Content-Length"])) + records.append((self.path, body, self.headers.get("Authorization"))) + if self.path == "/default-timeout/traces": + time.sleep(1.2) + elif delay and not self.path.endswith("metrics"): + time.sleep(delay) + self.send_response(200) + self.send_header("Content-Length", "0") + self.end_headers() + + def log_message(self, *args): + pass + + +def points(metrics, name): + for item in metrics: + for resource in item.get("resourceMetrics", []): + for scope in resource["scopeMetrics"]: + for metric in scope["metrics"]: + if metric["name"] == name: + yield from metric.get("sum", metric.get("histogram"))["dataPoints"] + + +def labels(point): + return {a["key"]: next(iter(a["value"].values())) for a in point["attributes"]} + + +async def injected_faults(endpoint, diagnostic): + from opentelemetry.sdk.trace import TracerProvider, Span as SDKSpan + from opentelemetry.sdk._logs import LoggerProvider + from opentelemetry.sdk.metrics import MeterProvider + from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter + from opentelemetry.exporter.otlp.proto.http._log_exporter import OTLPLogExporter + from opentelemetry.exporter.otlp.proto.http.metric_exporter import OTLPMetricExporter + + failure = RuntimeError(CANARY) + closed = [] + checks = [] + for signal, provider, method, exporter in ( + ("traces", TracerProvider, "get_tracer", OTLPSpanExporter), + ("logs", LoggerProvider, "get_logger", OTLPLogExporter), + ("metrics", MeterProvider, "get_meter", OTLPMetricExporter), + ): + for item in ("traces", "logs", "metrics"): + os.environ[f"OTEL_EXPORTER_OTLP_{item.upper()}_ENDPOINT"] = ( + endpoint + "/fault/" + item if item == signal else "" + ) + before = { + thread.ident for thread in threading.enumerate() if thread.name.startswith("Otel") + } + original_shutdown = exporter.shutdown + + def observed_shutdown(self, *args, **kwargs): + closed.append(signal) + return original_shutdown(self, *args, **kwargs) + + with redirect_stderr(diagnostic), patch.object(provider, method, side_effect=failure): + with patch.object(exporter, "shutdown", new=observed_shutdown): + telemetry.start_telemetry(enabled=True, service_version="fault-probe") + assert not telemetry.is_enabled() + assert telemetry.get_export_connection(signal) is None + assert closed[-1] == signal + assert { + thread.ident for thread in threading.enumerate() if thread.name.startswith("Otel") + } == before + checks.append(signal + "_initialization_rollback") + + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + endpoint + "/fault/" + signal.lower() + ) + with ( + redirect_stderr(diagnostic), + patch.object(TracerProvider, "get_tracer", side_effect=failure), + ): + telemetry.start_telemetry(enabled=True, service_version="fault-probe") + assert telemetry.is_enabled() and telemetry.get_export_connection("traces") is None + assert telemetry.get_export_connection("logs") and telemetry.get_export_connection( + "metrics" + ) + await telemetry.close_telemetry() + checks.append("failed_signal_not_published_with_other_signals_active") + + with redirect_stderr(diagnostic): + telemetry.start_telemetry(enabled=True, service_version="fault-probe") + runtime = telemetry._runtime + with patch.object(runtime.tracer, "start_span", side_effect=failure): + with telemetry.operation("ai.chat") as observation: + assert isinstance(observation.span, NonRecordingSpan) + business_result = 42 + assert business_result == 42 + checks.append("start_span_failure_isolated") + with patch.object(telemetry, "use_span", side_effect=failure): + with telemetry.operation("ai.chat"): + business_result = 43 + assert business_result == 43 + checks.append("context_activation_failure_isolated") + with ExitStack() as faults: + for target, method in ( + (runtime.logger, "emit"), + (runtime.tokens, "add"), + (runtime.missing_usage, "add"), + ): + faults.enter_context(patch.object(target, method, side_effect=failure)) + telemetry.emit_event("ai.finished", {}) + telemetry.record_ai_usage("chat", 1, None) + checks.append("log_and_usage_failure_isolated") + for business_error in (None, LookupError(CANARY), asyncio.CancelledError(CANARY)): + with ExitStack() as faults: + for target, method in ( + (SDKSpan, "set_attribute"), + (SDKSpan, "set_status"), + (SDKSpan, "end"), + (runtime.operations, "add"), + (runtime.duration, "record"), + ): + faults.enter_context(patch.object(target, method, side_effect=failure)) + try: + with telemetry.operation("ai.chat") as observation: + observation.outcome = "error" + if business_error is not None: + raise business_error + business_result = 44 + except BaseException as observed: + assert observed is business_error + else: + assert business_error is None and business_result == 44 + observation.span.end() + checks.append("finalization_preserves_result_exception_and_cancellation") + await telemetry.close_telemetry() + assert CANARY not in diagnostic.getvalue() + return checks + + +async def timeout_configuration(endpoint, diagnostic): + timeout_keys = ["OTEL_EXPORTER_OTLP_TIMEOUT"] + [ + f"OTEL_EXPORTER_OTLP_{signal}_TIMEOUT" for signal in ("TRACES", "LOGS", "METRICS") + ] + snapshots = {} + with patch.dict(os.environ), TemporaryDirectory() as temporary: + for key in timeout_keys: + os.environ.pop(key, None) + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + endpoint + "/timeout/" + signal.lower() + ) + + async def capture(label): + with redirect_stderr(diagnostic): + telemetry.start_telemetry(enabled=True, service_version="timeout-probe") + snapshots[label] = { + signal: connection[2] + if (connection := telemetry.get_export_connection(signal)) + else None + for signal in ("traces", "logs", "metrics") + } + with telemetry.operation("ai.chat"): + business_result = 42 + await telemetry.close_telemetry() + assert business_result == 42 + + os.environ["OTEL_EXPORTER_OTLP_TRACES_ENDPOINT"] = endpoint + "/default-timeout/traces" + before = len(records) + await capture("default") + assert snapshots["default"] == {"traces": 10, "logs": 10, "metrics": 10} + exported = [ + MessageToDict(ExportMetricsServiceRequest.FromString(body)) + for path, body, _ in records[before:] + if path.endswith("metrics") + ] + delivered = list(points(exported, "otel.sdk.exporter.span.exported")) + assert delivered and all("error.type" not in labels(point) for point in delivered) + snapshots["default_slow_receiver"] = { + "delay_seconds": 1.2, + "successful_spans": sum(int(point["asInt"]) for point in delivered), + } + os.environ["OTEL_EXPORTER_OTLP_TRACES_ENDPOINT"] = endpoint + "/timeout/traces" + dotenv = Path(temporary) / "probe.env" + dotenv.write_text( + "OTEL_EXPORTER_OTLP_TIMEOUT=2.5\nOTEL_EXPORTER_OTLP_TRACES_TIMEOUT=3.5\n" + ) + os.environ["INKCRE_ENV_FILE"] = str(dotenv) + os.environ["OTEL_EXPORTER_OTLP_TIMEOUT"] = "4" + os.environ["OTEL_EXPORTER_OTLP_METRICS_TIMEOUT"] = "6" + await capture("dotenv_signal_and_environment_common") + assert snapshots["dotenv_signal_and_environment_common"] == { + "traces": 3.5, + "logs": 4, + "metrics": 6, + } + os.environ["OTEL_EXPORTER_OTLP_TRACES_TIMEOUT"] = "7" + await capture("environment_signal_over_dotenv") + assert snapshots["environment_signal_over_dotenv"] == { + "traces": 7, + "logs": 4, + "metrics": 6, + } + for invalid in ("0", "-1", "31", "nan", "inf", "invalid"): + os.environ["OTEL_EXPORTER_OTLP_TRACES_TIMEOUT"] = invalid + await capture("invalid_" + invalid) + assert snapshots["invalid_" + invalid] == {"traces": None, "logs": 4, "metrics": 6} + os.environ["OTEL_EXPORTER_OTLP_TIMEOUT"] = "invalid" + os.environ["OTEL_EXPORTER_OTLP_TRACES_TIMEOUT"] = "30" + await capture("valid_signal_over_invalid_common") + assert snapshots["valid_signal_over_invalid_common"] == { + "traces": 30, + "logs": None, + "metrics": 6, + } + return snapshots + + +async def main(): + global delay + logging.getLogger().addHandler(logging.NullHandler()) + server = ThreadingHTTPServer(("127.0.0.1", 0), Receiver) + worker = threading.Thread(target=server.serve_forever, daemon=True) + worker.start() + endpoint = f"http://127.0.0.1:{server.server_port}" + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + f"{endpoint}/custom/{signal.lower()}" + ) + os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = ( + "Authorization=Bearer%20synthetic-private-write-token" + ) + os.environ["OTEL_RESOURCE_ATTRIBUTES"] = f"forbidden={CANARY}" + before_threads = {t.ident for t in threading.enumerate()} + telemetry.start_telemetry(enabled=False, service_version="probe") + with telemetry.operation("ai.chat") as observation: + assert isinstance(observation.span, NonRecordingSpan) + assert get_current_span() is observation.span + telemetry.emit_event("ai.finished", {"operation": "chat"}) + telemetry.record_ai_usage("chat", 1, 2) + parent = NonRecordingSpan(SpanContext(trace_id=123, span_id=456, is_remote=False)) + with use_span(parent): + with telemetry.operation("ai.chat") as observation: + assert get_current_span() is parent and observation.span is not parent + await telemetry.close_telemetry() + assert not telemetry.is_enabled() and records == [] + assert telemetry.get_export_connection("traces") is None + assert {t.ident for t in threading.enumerate()} == before_threads + + diagnostic = io.StringIO() + with redirect_stderr(diagnostic): + telemetry.start_telemetry( + enabled=True, + service_version="probe", + resource_attributes={"inkcre.deployment.id": "synthetic-deployment"}, + endpoints={"traces": endpoint + "/wrong-default"}, + ) + assert telemetry.is_enabled() + connection = telemetry.get_export_connection("traces") + assert connection is not None and connection[0] == endpoint + "/custom/traces" + connection[1]["authorization"] = "changed-copy" + assert telemetry.get_export_connection("traces")[1]["authorization"] != "changed-copy" + with telemetry.operation("ai.chat", attributes={"gen_ai.operation.name": "chat"}): + telemetry.emit_event("ai.finished", {"gen_ai.usage.input_tokens": 3}) + telemetry.record_ai_usage("chat", 3, None) + logging.getLogger("inkcre").warning(CANARY) + logging.getLogger("opentelemetry.trace.span").warning(CANARY) + TraceState.from_header([CANARY]) + try: + with telemetry.operation("ai.chat") as observation: + observation.outcome = "cancelled" + raise RuntimeError(CANARY) + except RuntimeError: + pass + try: + with telemetry.operation("ai.chat") as observation: + observation.outcome = "error" + raise asyncio.CancelledError(CANARY) + except asyncio.CancelledError: + pass + with telemetry.operation("agent.tool") as observation: + observation.outcome = "error" + telemetry.record_ai_usage("embeddings", 0, -1) + started = time.monotonic() + await telemetry.close_telemetry() + healthy_close = time.monotonic() - started + assert CANARY not in diagnostic.getvalue() + assert healthy_close < 5, healthy_close + assert {r[0] for r in records} == {"/custom/traces", "/custom/logs", "/custom/metrics"} + assert all(CANARY.encode() not in r[1] for r in records) + assert all(r[2] == "Bearer synthetic-private-write-token" for r in records) + decoded = {"traces": [], "logs": [], "metrics": []} + classes = { + "traces": ExportTraceServiceRequest, + "logs": ExportLogsServiceRequest, + "metrics": ExportMetricsServiceRequest, + } + for path, data, _ in records: + signal = path.rsplit("/", 1)[-1] + decoded[signal].append(MessageToDict(classes[signal].FromString(data))) + spans = [ + span + for req in decoded["traces"] + for rs in req["resourceSpans"] + for scope in rs["scopeSpans"] + for span in scope["spans"] + ] + assert len(spans) == 4 + assert all( + not span.get("events") and not span.get("status", {}).get("message") for span in spans + ) + logs = [ + log + for req in decoded["logs"] + for rs in req["resourceLogs"] + for scope in rs["scopeLogs"] + for log in scope["logRecords"] + ] + assert len(logs) == 1 and logs[0]["body"]["stringValue"] == "ai.finished" + assert logs[0]["traceId"] in {span["traceId"] for span in spans} + token_points = list(points(decoded["metrics"], "inkcre.ai.token.usage")) + missing_points = list(points(decoded["metrics"], "inkcre.ai.usage.missing")) + assert len(token_points) == 2 and sorted(int(p["asInt"]) for p in token_points) == [0, 3] + assert sum(int(p["asInt"]) for p in missing_points) == 2 + duration_points = list(points(decoded["metrics"], "inkcre.operation.duration")) + assert duration_points and all( + point["explicitBounds"][0] <= 0.005 and point["explicitBounds"][-1] >= 300 + for point in duration_points + ), "Duration buckets must cover millisecond HTTP calls and multi-minute Jobs" + operation_points = list(points(decoded["metrics"], "inkcre.operation.count")) + assert {labels(p)["outcome"] for p in operation_points} == { + "success", + "error", + "cancelled", + } + assert sum(int(p["asInt"]) for p in operation_points) == 4 + assert all(set(labels(p)) == {"operation", "outcome"} for p in operation_points) + + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + f"http://user:{CANARY}@localhost/ingest" + ) + with redirect_stderr(diagnostic): + telemetry.start_telemetry(enabled=True, service_version="probe") + assert not telemetry.is_enabled() and CANARY not in diagnostic.getvalue() + + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + f"{endpoint}/slow/{signal.lower()}" + ) + # Invalid secret headers are rejected without printing their raw value. + os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = CANARY + with redirect_stderr(diagnostic): + telemetry.start_telemetry(enabled=True, service_version="probe") + assert not telemetry.is_enabled() and CANARY not in diagnostic.getvalue() + os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = "Authorization=synthetic" + + # A single configured signal does not initialize the other signal workers. + os.environ["OTEL_EXPORTER_OTLP_LOGS_ENDPOINT"] = "" + os.environ["OTEL_EXPORTER_OTLP_METRICS_ENDPOINT"] = "" + count = len(records) + with redirect_stderr(diagnostic): + telemetry.start_telemetry(enabled=True, service_version="probe") + with telemetry.operation("ai.chat"): + telemetry.emit_event("ai.finished", {}) + telemetry.record_ai_usage("chat", 1, 1) + await telemetry.close_telemetry() + assert len(records) == count + 1 and records[-1][0].endswith("traces") + + # A real refused connection must not change the operation result. + unused = ThreadingHTTPServer(("127.0.0.1", 0), Receiver) + refused = f"http://127.0.0.1:{unused.server_port}/ingest" + unused.server_close() + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = refused + with redirect_stderr(diagnostic): + telemetry.start_telemetry(enabled=True, service_version="probe") + with telemetry.operation("ai.chat"): + result = 42 + telemetry.emit_event("ai.finished", {}) + telemetry.record_ai_usage("chat", 1, None) + started = time.monotonic() + await telemetry.close_telemetry() + refused_close = time.monotonic() - started + assert result == 42 and refused_close < 5 + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + f"{endpoint}/slow/{signal.lower()}" + ) + delay = 2.0 + before = len(records) + with redirect_stderr(diagnostic): + telemetry.start_telemetry(enabled=True, service_version="probe") + started = time.monotonic() + for _ in range(1024): + with telemetry.operation("ai.chat"): + telemetry.emit_event("ai.finished", {}) + telemetry.record_ai_usage("chat", 1, 1) + operation_elapsed = time.monotonic() - started + started = time.monotonic() + await telemetry.close_telemetry() + slow_close = time.monotonic() - started + assert operation_elapsed < 1, operation_elapsed + assert slow_close < 5, slow_close + assert not telemetry.is_enabled() and len(records) > before + assert CANARY not in diagnostic.getvalue() + pipeline = [] + dropped = {} + export_failures = {} + for path, body, _ in records[before:]: + if not path.endswith("metrics"): + continue + batch = ExportMetricsServiceRequest.FromString(body) + for resource in batch.resource_metrics: + for scope in resource.scope_metrics: + if scope.scope.name != "opentelemetry-sdk": + continue + for metric in scope.metrics: + pipeline.append(metric.name) + data = getattr(metric, metric.WhichOneof("data")) + for point in data.data_points: + attrs = {attr.key: attr.value for attr in point.attributes} + assert set(attrs) <= { + "otel.component.type", + "error.type", + "http.response.status_code", + } + assert not point.exemplars + if attrs.get("error.type") and attrs["error.type"].string_value == "queue_full": + dropped[metric.name] = dropped.get(metric.name, 0) + point.as_int + if ( + metric.name.startswith("otel.sdk.exporter.") + and metric.name.endswith(".exported") + and attrs.get("error.type") + ): + export_failures[metric.name] = ( + export_failures.get(metric.name, 0) + point.as_int + ) + assert dropped.get("otel.sdk.processor.span.processed", 0) > 0, dropped + assert dropped.get("otel.sdk.processor.log.processed", 0) > 0, dropped + assert export_failures.get("otel.sdk.exporter.span.exported", 0) > 0, export_failures + assert export_failures.get("otel.sdk.exporter.log.exported", 0) > 0, export_failures + assert "otel.sdk.exporter.operation.duration" in pipeline, pipeline + assert "otel.sdk.processor.span.queue.capacity" in pipeline + assert "otel.sdk.processor.log.queue.size" in pipeline + delay = 0 + fault_checks = await injected_faults(endpoint, diagnostic) + timeout_checks = await timeout_configuration(endpoint, diagnostic) + server.shutdown() + server.server_close() + evidence = { + "off_no_threads_or_exports": True, + "three_signals": True, + "canary_absent": True, + "usage_known_zero_missing": True, + "exception_and_cancellation_propagate": True, + "restart": True, + "invalid_configuration_isolated": True, + "healthy_shutdown_seconds": healthy_close, + "slow_receiver_backlog_shutdown_seconds": slow_close, + "refused_receiver_shutdown_seconds": refused_close, + "backlog_business_seconds": operation_elapsed, + "requests": len(records), + "native_pipeline_metric_names": sorted(set(pipeline)), + "native_queue_drops": dropped, + "native_export_failures": export_failures, + "native_metric_attributes_and_exemplars_safe": True, + "fault_injection_checks": fault_checks, + "timeout_configuration": timeout_checks, + } + destination = Path(__file__).parent / "evidence" / "foundation-probe.json" + destination.write_text(json.dumps(evidence, indent=2) + "\n") + print(json.dumps(evidence, indent=2)) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/tasks/observability-foundation/experiments/implementation-db.py b/tasks/observability-foundation/experiments/implementation-db.py new file mode 100644 index 0000000..d2f9b46 --- /dev/null +++ b/tasks/observability-foundation/experiments/implementation-db.py @@ -0,0 +1,51 @@ +"""Run source lifecycle commands only against this task's disposable database.""" + +import json +import os +from pathlib import Path +import subprocess +import sys + +import psycopg +from psycopg import sql + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +CREDENTIALS = json.loads((HERE / "runtime/database-credential.json").read_text()) +DB = "o11y_impl" +ADMIN = f"postgresql://postgres:{CREDENTIALS['admin']}@127.0.0.1:35432/{DB}" +CORE = f"postgresql+psycopg://inkcre_core:{CREDENTIALS['core']}@127.0.0.1:35432/{DB}" + + +def environment(): + return { + **os.environ, + "INKCRE_ENV_FILE": "", + "DATABASE_URL": CORE, + "MIGRATION_DATABASE_URL": ADMIN, + "JWT_SECRET": CREDENTIALS["jwt"], + "CORE_DATABASE_PASSWORD": CREDENTIALS["core"], + "POSTGREST_DATABASE_PASSWORD": CREDENTIALS["rest"], + } + + +def main(): + if sys.argv[1:] == ["create"]: + with psycopg.connect(ADMIN.rsplit("/", 1)[0] + "/postgres", autocommit=True) as conn: + if not conn.execute("SELECT 1 FROM pg_database WHERE datname=%s", (DB,)).fetchone(): + conn.execute(sql.SQL("CREATE DATABASE {}").format(sql.Identifier(DB))) + print("Disposable implementation database is available; old synthetic data retained.") + return + result = subprocess.run( # noqa: S603 + sys.argv[1:], cwd=ROOT, env=environment(), capture_output=True, text=True, check=False + ) + output = result.stdout + result.stderr + (HERE / "runtime/implementation-db-last.log").write_text(output) + for value in CREDENTIALS.values(): + output = output.replace(value, "") + print(output, end="") + raise SystemExit(result.returncode) + + +if __name__ == "__main__": + main() diff --git a/tasks/observability-foundation/experiments/lab.py b/tasks/observability-foundation/experiments/lab.py new file mode 100644 index 0000000..451ae2d --- /dev/null +++ b/tasks/observability-foundation/experiments/lab.py @@ -0,0 +1,150 @@ +"""Task-only isolated Docker recipe; python3 lab.py up|stop|stats|logs|backend- +stop|backend-start. + +Uses the already-authorized SSH Docker host, never the SVC database project. +Stop preserves synthetic data for repeatable inspection; removal is a separate explicit +command. +""" + +import base64 +import json +import os +from pathlib import Path +import secrets +import shlex +import subprocess +import sys + + +HERE = Path(__file__).resolve().parent +RUNTIME = HERE / "runtime" +PROJECT = "inkcre-o11y-g1-b0a97f7c" +HOST = "wsl.win-ws.localhost" +DOCKER = "/mnt/c/Program Files/Docker/Docker/resources/bin/docker.exe" +SOCKET = "/tmp/inkcre-o11y-g1-b0a97f7c.sock" +IMAGES = { + "openobserve": "openobserve/openobserve@sha256:d4a878fac1f6c56003764f7f2a1625668917388f167e222c8c810de3f54c56ba", # noqa: E501 + "collector": "otel/opentelemetry-collector@sha256:310a800ad69ee430e7c541796852a242c9c7db97aaad4daa5ccf843c525fbdb2", # noqa: E501 +} + + +def docker(*args, data=None): + command = [DOCKER, *args] + if os.getenv("LAB_WSL_INTEROP"): + command = ["env", "WSL_INTEROP=" + os.environ["LAB_WSL_INTEROP"], *command] + return subprocess.check_output( # noqa: S603 + ["ssh", "-o", "BatchMode=yes", "-o", "ConnectTimeout=8", HOST, shlex.join(command)], + input=data, + text=True, + timeout=60, + ) + + +def credentials(): + return json.loads((RUNTIME / "credential.json").read_text()) + + +def compose(): + cred = credentials() + auth = base64.b64encode(f"{cred['email']}:{cred['password']}".encode()).decode() + return json.dumps( + { + "services": { + "openobserve": { + "image": IMAGES["openobserve"], + "cpus": 1.5, + "mem_limit": "3584m", + "ports": ["127.0.0.1:35080:5080"], + "environment": { + "ZO_ROOT_USER_EMAIL": cred["email"], + "ZO_ROOT_USER_PASSWORD": cred["password"], + "ZO_DATA_DIR": "/data", + "ZO_LOCAL_MODE": "true", + "ZO_TELEMETRY": "false", + "ZO_PROMETHEUS_ENABLED": "true", + "ZO_DISK_CACHE_MAX_SIZE": "256", + }, + "volumes": ["data:/data"], + }, + "collector": { + "image": IMAGES["collector"], + "cpus": 0.5, + "mem_limit": "512m", + "ports": ["127.0.0.1:34318:4318", "127.0.0.1:38888:8888"], + "command": ["--config=/etc/otelcol/lab.yaml"], + "environment": {"LAB_AUTH": auth}, + "configs": [{"source": "collector", "target": "/etc/otelcol/lab.yaml"}], + }, + }, + # Compose expands dollar signs; the Collector owns this env expansion. + "configs": { + "collector": {"content": (HERE / "collector.yaml").read_text().replace("$", "$$")} + }, + "volumes": {"data": {}}, + "networks": {"default": {}}, + } + ) + + +def main(action): + if action == "up": + RUNTIME.mkdir(exist_ok=True, mode=0o700) + credential = RUNTIME / "credential.json" + if not credential.exists(): + with os.fdopen( + os.open(credential, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600), "w" + ) as file: + json.dump( + {"email": "lab@example.invalid", "password": secrets.token_urlsafe(32)}, file + ) + print(docker("compose", "-p", PROJECT, "-f", "-", "up", "-d", data=compose())) + if not Path(SOCKET).exists(): + subprocess.run( # noqa: S603 + [ + "ssh", + "-M", + "-S", + SOCKET, + "-fNT", + "-o", + "BatchMode=yes", + "-o", + "ExitOnForwardFailure=yes", + "-L", + "127.0.0.1:35080:127.0.0.1:35080", + "-L", + "127.0.0.1:34318:127.0.0.1:34318", + "-L", + "127.0.0.1:38888:127.0.0.1:38888", + HOST, + ], + check=True, + ) + elif action in {"stop", "remove"}: + command = ["stop"] if action == "stop" else ["down", "--volumes"] + print(docker("compose", "-p", PROJECT, "-f", "-", *command, data=compose())) + if Path(SOCKET).exists(): + subprocess.run(["ssh", "-S", SOCKET, "-O", "exit", HOST], check=True) # noqa: S603 + elif action == "stats": + print( + docker( + "stats", + "--no-stream", + "--format", + "{{json .}}", + f"{PROJECT}-openobserve-1", + f"{PROJECT}-collector-1", + ) + ) + elif action == "logs": + print( + docker("compose", "-p", PROJECT, "-f", "-", "logs", "--tail", "40", data=compose()) + ) + elif action in {"backend-stop", "backend-start"}: + print(docker(action.removeprefix("backend-"), f"{PROJECT}-openobserve-1")) + else: + raise SystemExit("Use up, stop, remove, stats, logs, backend-stop, or backend-start") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tasks/observability-foundation/experiments/legacy-consumers.mjs b/tasks/observability-foundation/experiments/legacy-consumers.mjs new file mode 100644 index 0000000..7c8a4a8 --- /dev/null +++ b/tasks/observability-foundation/experiments/legacy-consumers.mjs @@ -0,0 +1,29 @@ +import assert from 'node:assert/strict'; +import { createRequire } from 'node:module'; +import { resolve } from 'node:path'; + +const webRoot = resolve(process.argv[2]); +const { build } = createRequire(resolve(webRoot, 'package.json'))( + resolve(webRoot, 'node_modules/.pnpm/esbuild@0.28.1/node_modules/esbuild/lib/main.js') +); +const bundle = await build({ + entryPoints: [resolve(webRoot, 'packages/core/src/job/job.ts')], + bundle: true, write: false, platform: 'node', format: 'esm', + plugins: [{ + name: 'no-database-io', + setup(build) { + build.onResolve({filter: /\/base\/db-api$/}, () => ({path: 'db-api', namespace: 'probe'})); + build.onLoad({filter: /.*/, namespace: 'probe'}, () => ({contents: 'export class DBAPIClient {}'})); + }, + }], +}); +// Bundle the actual old schema. Only its I/O dependency is replaced; no live database is used. +const { Job } = await import(`data:text/javascript;base64,${Buffer.from(bundle.outputFiles[0].text).toString('base64')}`); +const base = {id: 42, type: 'synthetic', parameters: {}, state: {}, timeout_seconds: 30, status: 'pending', created_at: '2026-10-01T00:00:00Z', started_at: null, closed_at: null}; +for (const carrier of [{}, {submission_traceparent: null, submission_tracestate: null}, {submission_traceparent: '00-' + '1'.repeat(32) + '-' + '2'.repeat(16) + '-01', submission_tracestate: 'inkcre=synthetic'}]) { + const job = Job.parse({...base, ...carrier}); + assert.equal(job.id, 42); + assert.equal(job.status, 'pending'); + assert.equal(job.submission_traceparent, undefined); +} +console.log(JSON.stringify({actual_legacy_ts_schema: 'accepted missing, NULL and additional carrier fields; unknown fields stripped', database_updates: 'not exercised'}, null, 2)); diff --git a/tasks/observability-foundation/experiments/legacy-consumers.py b/tasks/observability-foundation/experiments/legacy-consumers.py new file mode 100644 index 0000000..b8f9802 --- /dev/null +++ b/tasks/observability-foundation/experiments/legacy-consumers.py @@ -0,0 +1,46 @@ +"""Run through the core-py PDM environment against its unchanged current schema.""" + +import json +from pathlib import Path +import sys + +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) +from app.schemas.job import JobCreateForm, JobModel +from pydantic import ValidationError + +base = dict( + id=42, + type="synthetic", + parameters={}, + state={}, + timeout_seconds=30, + status="pending", + created_at="2026-10-01T00:00:00Z", +) +for carrier in [ + {}, + {"submission_traceparent": None, "submission_tracestate": None}, + { + "submission_traceparent": "00-" + "1" * 32 + "-" + "2" * 16 + "-01", + "submission_tracestate": "inkcre=synthetic", + }, +]: + job = JobModel.model_validate({**base, **carrier}) + assert job.id == 42 and job.status == "pending" + assert "submission_traceparent" not in job.model_dump() +try: + JobCreateForm.model_validate({"type": "synthetic", "submission_traceparent": "synthetic"}) +except ValidationError: + pass +else: + raise AssertionError("Old REST creation input unexpectedly admits telemetry fields") +print( + json.dumps( + { + "actual_legacy_python_schema": "accepted missing, NULL and additional carrier fields; unknown fields ignored", # noqa: E501 + "old_rest_create_form": "rejects extra carrier fields; use HTTP propagation at the transport boundary", # noqa: E501 + "database_updates": "not exercised", + }, + indent=2, + ) +) diff --git a/tasks/observability-foundation/experiments/migration-roundtrip.py b/tasks/observability-foundation/experiments/migration-roundtrip.py new file mode 100644 index 0000000..b1c5554 --- /dev/null +++ b/tasks/observability-foundation/experiments/migration-roundtrip.py @@ -0,0 +1,365 @@ +"""Exercise only the carrier migration on one fresh task-owned PostgreSQL database.""" + +from __future__ import annotations + +from contextlib import redirect_stderr, redirect_stdout +import io +import json +import os +from pathlib import Path +import subprocess +import sys + +from alembic import command +from alembic.config import Config +import psycopg +from psycopg import sql + + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +REPORT_PATH = HERE / "evidence/migration-roundtrip.json" +DATABASE = "o11y_migration_roundtrip" +BASE_REVISION = "a0465e3b028f" +TARGET_REVISION = "3d9593b0c855" +JOB_TYPE = "o11y-migration-roundtrip" +CARRIER_COLUMNS = ("submission_traceparent", "submission_tracestate") +CARRIER_CONSTRAINTS = { + "submission_traceparent": "jobs_submission_traceparent_capacity", + "submission_tracestate": "jobs_submission_tracestate_capacity", +} +LEGACY_LOG_BODY = "o11y-migration-roundtrip-legacy-log" +LEGACY_LOG_TRACE_ID = "legacy.migration.roundtrip" + + +def _credentials() -> dict[str, str]: + return json.loads((HERE / "runtime/database-credential.json").read_text()) + + +def _urls(credentials: dict[str, str]) -> tuple[str, str, str]: + admin_server = f"postgresql://postgres:{credentials['admin']}@127.0.0.1:35432/postgres" + admin_database = admin_server.rsplit("/", 1)[0] + f"/{DATABASE}" + core_database = ( + f"postgresql+psycopg://inkcre_core:{credentials['core']}@127.0.0.1:35432/{DATABASE}" + ) + return admin_server, admin_database, core_database + + +def _environment( + credentials: dict[str, str], + admin_database: str, + core_database: str, +) -> dict[str, str]: + return { + **os.environ, + "INKCRE_ENV_FILE": "", + "DATABASE_URL": core_database, + "MIGRATION_DATABASE_URL": admin_database, + "JWT_SECRET": credentials["jwt"], + "CORE_DATABASE_PASSWORD": credentials["core"], + "POSTGREST_DATABASE_PASSWORD": credentials["rest"], + } + + +def _write_report(report: dict[str, object]) -> None: + REPORT_PATH.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") + + +def _create_fresh_database(admin_server: str) -> None: + with psycopg.connect(admin_server, autocommit=True) as connection: + exists = connection.execute( + "SELECT 1 FROM pg_database WHERE datname = %s", (DATABASE,) + ).fetchone() + if exists is not None: + raise RuntimeError("roundtrip_database_already_exists") + connection.execute(sql.SQL("CREATE DATABASE {}").format(sql.Identifier(DATABASE))) + + +def _initialize_runtime_database(environment: dict[str, str]) -> None: + result = subprocess.run( + [sys.executable, "scripts/database.py", "init", "--profile", "runtime"], + cwd=ROOT, + env=environment, + capture_output=True, + check=False, + text=True, + ) + if result.returncode != 0: + raise RuntimeError("runtime_database_initialization_failed") + + +def _migrate(direction: str, revision: str) -> None: + config = Config(ROOT / "alembic.ini") + sink = io.StringIO() + try: + with redirect_stdout(sink), redirect_stderr(sink): + if direction == "upgrade": + command.upgrade(config, revision) + else: + command.downgrade(config, revision) + except Exception as error: + raise RuntimeError(f"alembic_{direction}_failed") from error + + +def _heads(connection: psycopg.Connection) -> tuple[str, ...]: + return tuple( + row[0] + for row in connection.execute( + "SELECT version_num FROM public.alembic_version ORDER BY version_num" + ).fetchall() + ) + + +def _columns(connection: psycopg.Connection) -> set[str]: + return { + row[0] + for row in connection.execute( + """ + SELECT column_name + FROM information_schema.columns + WHERE table_schema = 'inkcre' AND table_name = 'jobs' + """ + ).fetchall() + } + + +def _constraints(connection: psycopg.Connection) -> dict[str, str]: + names = tuple(CARRIER_CONSTRAINTS.values()) + return dict( + connection.execute( + """ + SELECT conname, pg_get_constraintdef(oid) + FROM pg_constraint + WHERE conrelid = 'inkcre.jobs'::regclass AND conname = ANY(%s) + """, + (list(names),), + ).fetchall() + ) + + +def _assert_head(connection: psycopg.Connection, revision: str) -> None: + if _heads(connection) != (revision,): + raise AssertionError("unexpected_migration_head") + + +def _assert_carrier_schema(connection: psycopg.Connection) -> None: + if not set(CARRIER_COLUMNS) <= _columns(connection): + raise AssertionError("carrier_columns_missing") + definitions = _constraints(connection) + if set(definitions) != set(CARRIER_CONSTRAINTS.values()): + raise AssertionError("carrier_capacity_constraints_missing") + if not all("octet_length" in value and "512" in value for value in definitions.values()): + raise AssertionError("carrier_capacity_constraint_definition_unexpected") + + +def _insert_legacy_rows(connection: psycopg.Connection) -> tuple[int, int]: + connection.execute( + """ + INSERT INTO inkcre.job_types + (id, description, parameters_schema, default_timeout_seconds) + VALUES (%s, 'Synthetic legacy migration Job', '{}'::jsonb, 60) + """, + (JOB_TYPE,), + ) + job_id = connection.execute( + """ + INSERT INTO inkcre.jobs (type, parameters, state, timeout_seconds) + VALUES (%s, '{}'::jsonb, '{}'::jsonb, 60) + RETURNING id + """, + (JOB_TYPE,), + ).fetchone()[0] + log_id = connection.execute( + """ + INSERT INTO inkcre.logs (severity_number, severity_text, body, trace_id, attributes) + VALUES (9, 'INFO', %s, %s, '{"probe":"migration-roundtrip"}'::jsonb) + RETURNING id + """, + (LEGACY_LOG_BODY, LEGACY_LOG_TRACE_ID), + ).fetchone()[0] + return job_id, log_id + + +def _legacy_rows_preserved( + connection: psycopg.Connection, job_id: int, log_id: int +) -> bool: + job = connection.execute( + """ + SELECT id, type, parameters, state, timeout_seconds, status + FROM inkcre.jobs + WHERE id = %s + """, + (job_id,), + ).fetchone() + log = connection.execute( + "SELECT id, body, trace_id, attributes FROM inkcre.logs WHERE id = %s", (log_id,) + ).fetchone() + return job == (job_id, JOB_TYPE, {}, {}, 60, "pending") and log == ( + log_id, + LEGACY_LOG_BODY, + LEGACY_LOG_TRACE_ID, + {"probe": "migration-roundtrip"}, + ) + + +def _insert_maximum_carrier(connection: psycopg.Connection) -> int: + value = "x" * 512 + row = connection.execute( + """ + INSERT INTO inkcre.jobs + ( + type, + parameters, + state, + timeout_seconds, + submission_traceparent, + submission_tracestate + ) + VALUES (%s, '{}'::jsonb, '{}'::jsonb, 60, %s, %s) + RETURNING id + """, + (JOB_TYPE, value, value), + ).fetchone() + lengths = connection.execute( + """ + SELECT octet_length(submission_traceparent), octet_length(submission_tracestate) + FROM inkcre.jobs + WHERE id = %s + """, + (row[0],), + ).fetchone() + if lengths != (512, 512): + raise AssertionError("maximum_carrier_not_preserved") + return row[0] + + +def _oversized_carriers_rejected(connection: psycopg.Connection) -> list[str]: + rejected = [] + for column in CARRIER_COLUMNS: + statement = sql.SQL( + """ + INSERT INTO inkcre.jobs (type, parameters, state, timeout_seconds, {}) + VALUES (%s, '{{}}'::jsonb, '{{}}'::jsonb, 60, %s) + """ + ).format(sql.Identifier(column)) + try: + connection.execute(statement, (JOB_TYPE, "x" * 513)) + except psycopg.errors.CheckViolation: + rejected.append(column) + else: + raise AssertionError("oversized_carrier_accepted") + return rejected + + +def _carrier_values( + connection: psycopg.Connection, job_id: int +) -> tuple[str | None, str | None]: + return connection.execute( + """ + SELECT submission_traceparent, submission_tracestate + FROM inkcre.jobs + WHERE id = %s + """, + (job_id,), + ).fetchone() + + +def main() -> int: + stage = "load_credentials" + report: dict[str, object] = {"database": DATABASE, "status": "failed"} + try: + credentials = _credentials() + admin_server, admin_database, core_database = _urls(credentials) + environment = _environment(credentials, admin_database, core_database) + os.environ.update(environment) + + stage = "create_fresh_database" + _create_fresh_database(admin_server) + + stage = "initialize_runtime_database" + _initialize_runtime_database(environment) + with psycopg.connect(admin_database, autocommit=True) as connection: + _assert_head(connection, TARGET_REVISION) + + stage = "downgrade_to_base" + _migrate("downgrade", BASE_REVISION) + with psycopg.connect(admin_database, autocommit=True) as connection: + _assert_head(connection, BASE_REVISION) + if set(CARRIER_COLUMNS) & _columns(connection): + raise AssertionError("carrier_columns_present_after_downgrade") + if _constraints(connection): + raise AssertionError("carrier_constraints_present_after_downgrade") + legacy_job_id, legacy_log_id = _insert_legacy_rows(connection) + if not _legacy_rows_preserved(connection, legacy_job_id, legacy_log_id): + raise AssertionError("legacy_rows_not_written") + + stage = "upgrade_to_target" + _migrate("upgrade", TARGET_REVISION) + with psycopg.connect(admin_database, autocommit=True) as connection: + _assert_head(connection, TARGET_REVISION) + _assert_carrier_schema(connection) + if _carrier_values(connection, legacy_job_id) != (None, None): + raise AssertionError("legacy_carrier_defaults_not_null") + legacy_after_first_upgrade = _legacy_rows_preserved( + connection, legacy_job_id, legacy_log_id + ) + maximum_carrier_job_id = _insert_maximum_carrier(connection) + first_capacity_rejections = _oversized_carriers_rejected(connection) + + stage = "downgrade_roundtrip" + _migrate("downgrade", BASE_REVISION) + with psycopg.connect(admin_database, autocommit=True) as connection: + _assert_head(connection, BASE_REVISION) + if set(CARRIER_COLUMNS) & _columns(connection): + raise AssertionError("carrier_columns_survived_downgrade") + legacy_after_downgrade = _legacy_rows_preserved( + connection, legacy_job_id, legacy_log_id + ) + + stage = "reupgrade_to_target" + _migrate("upgrade", TARGET_REVISION) + with psycopg.connect(admin_database, autocommit=True) as connection: + _assert_head(connection, TARGET_REVISION) + _assert_carrier_schema(connection) + if _carrier_values(connection, maximum_carrier_job_id) != (None, None): + raise AssertionError("downgrade_did_not_remove_carrier") + legacy_after_reupgrade = _legacy_rows_preserved( + connection, legacy_job_id, legacy_log_id + ) + reupgrade_capacity_rejections = _oversized_carriers_rejected(connection) + + report = { + "database": DATABASE, + "status": "passed", + "initial_head": TARGET_REVISION, + "downgrade_head": BASE_REVISION, + "reupgrade_head": TARGET_REVISION, + "legacy_job_id": legacy_job_id, + "legacy_pg_log_id": legacy_log_id, + "legacy_job_and_pg_log_preserved": { + "after_first_upgrade": legacy_after_first_upgrade, + "after_downgrade": legacy_after_downgrade, + "after_reupgrade": legacy_after_reupgrade, + }, + "carrier_columns_and_capacity_checks_present": True, + "accepted_512_byte_carriers": True, + "rejected_513_byte_carriers": { + "after_first_upgrade": first_capacity_rejections, + "after_reupgrade": reupgrade_capacity_rejections, + }, + "downgrade_removes_carriers_as_expected": True, + "reupgrade_restores_columns_with_null_legacy_values": True, + } + except Exception as error: + report = { + **report, + "stage": stage, + "error_type": type(error).__name__, + } + _write_report(report) + print(json.dumps(report, indent=2, sort_keys=True)) + return 0 if report["status"] == "passed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tasks/observability-foundation/experiments/package-lock.json b/tasks/observability-foundation/experiments/package-lock.json new file mode 100644 index 0000000..2110400 --- /dev/null +++ b/tasks/observability-foundation/experiments/package-lock.json @@ -0,0 +1,99 @@ +{ + "name": "inkcre-observability-contract-probe", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "inkcre-observability-contract-probe", + "dependencies": { + "@opentelemetry/api": "1.9.1", + "@opentelemetry/core": "2.11.0", + "@opentelemetry/sdk-trace-base": "2.11.0" + } + }, + "node_modules/@opentelemetry/api": { + "version": "1.9.1", + "resolved": "https://registry.npmjs.org/@opentelemetry/api/-/api-1.9.1.tgz", + "integrity": "sha512-gLyJlPHPZYdAk1JENA9LeHejZe1Ti77/pTeFm/nMXmQH/HFZlcS/O2XJB+L8fkbrNSqhdtlvjBVjxwUYanNH5Q==", + "license": "Apache-2.0", + "engines": { + "node": ">=8.0.0" + } + }, + "node_modules/@opentelemetry/core": { + "version": "2.11.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/core/-/core-2.11.0.tgz", + "integrity": "sha512-7YP44XH0tV6+Mb54x2YGf84i7yi+31MBZlE8JwvozkxyTvXbSp10X7cI7YE49ChJ3shMJoBmCJF3+1QFBJctGA==", + "license": "Apache-2.0", + "dependencies": { + "@opentelemetry/semantic-conventions": "^1.29.0" + }, + "engines": { + "node": "^18.19.0 || >=20.6.0" + }, + "peerDependencies": { + "@opentelemetry/api": ">=1.0.0 <1.10.0" + } + }, + "node_modules/@opentelemetry/resources": { + "version": "2.11.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/resources/-/resources-2.11.0.tgz", + "integrity": "sha512-Ie7+8q8MDF4FAEQCKVMTx3ReUvxiIAgIiiW3c9JdmP8+HMcDy20puT+AHjexnExgnbvBxjQ9fjkFDWrikJ2jQA==", + "license": "Apache-2.0", + "dependencies": { + "@opentelemetry/core": "2.11.0", + "@opentelemetry/semantic-conventions": "^1.29.0" + }, + "engines": { + "node": "^18.19.0 || >=20.6.0" + }, + "peerDependencies": { + "@opentelemetry/api": ">=1.3.0 <1.10.0" + } + }, + "node_modules/@opentelemetry/sdk-trace": { + "version": "2.11.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-trace/-/sdk-trace-2.11.0.tgz", + "integrity": "sha512-fFnTqGm8/G73GQVnxYi7LXa1ZVYEUvgL6XI1LpvV0bPC7WQ/ZGgKxCSl8FnlZBKto9JHHEFTO6s6CUpvvtwFrA==", + "license": "Apache-2.0", + "dependencies": { + "@opentelemetry/core": "2.11.0", + "@opentelemetry/resources": "2.11.0", + "@opentelemetry/semantic-conventions": "^1.29.0" + }, + "engines": { + "node": "^18.19.0 || >=20.6.0" + }, + "peerDependencies": { + "@opentelemetry/api": ">=1.3.0 <1.10.0" + } + }, + "node_modules/@opentelemetry/sdk-trace-base": { + "version": "2.11.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/sdk-trace-base/-/sdk-trace-base-2.11.0.tgz", + "integrity": "sha512-H19x/TX/LZdqiYOjM7fqtSxwlplC5pgelavqbQdHbhdq0q/AI/TGkM2dfGuuynTXmJPeF2HoZVoPDu+TGoW78A==", + "license": "Apache-2.0", + "dependencies": { + "@opentelemetry/core": "2.11.0", + "@opentelemetry/resources": "2.11.0", + "@opentelemetry/sdk-trace": "2.11.0", + "@opentelemetry/semantic-conventions": "^1.29.0" + }, + "engines": { + "node": "^18.19.0 || >=20.6.0" + }, + "peerDependencies": { + "@opentelemetry/api": ">=1.3.0 <1.10.0" + } + }, + "node_modules/@opentelemetry/semantic-conventions": { + "version": "1.43.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/semantic-conventions/-/semantic-conventions-1.43.0.tgz", + "integrity": "sha512-eSYWTm620tTk45EKSedaUL8MFYI8hW164hIXsgIHyxu3VobUB3fFCu5t0hQby6OoWRPsG1KkKUG2M5UadiLiVg==", + "license": "Apache-2.0", + "engines": { + "node": ">=14" + } + } + } +} diff --git a/tasks/observability-foundation/experiments/package.json b/tasks/observability-foundation/experiments/package.json new file mode 100644 index 0000000..18a743e --- /dev/null +++ b/tasks/observability-foundation/experiments/package.json @@ -0,0 +1,10 @@ +{ + "name": "inkcre-observability-contract-probe", + "private": true, + "type": "module", + "dependencies": { + "@opentelemetry/api": "1.9.1", + "@opentelemetry/core": "2.11.0", + "@opentelemetry/sdk-trace-base": "2.11.0" + } +} diff --git a/tasks/observability-foundation/experiments/propagation.mjs b/tasks/observability-foundation/experiments/propagation.mjs new file mode 100644 index 0000000..df8d4d8 --- /dev/null +++ b/tasks/observability-foundation/experiments/propagation.mjs @@ -0,0 +1,36 @@ +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import { ROOT_CONTEXT, trace, defaultTextMapGetter, defaultTextMapSetter } from '@opentelemetry/api'; +import { W3CTraceContextPropagator } from '@opentelemetry/core'; +import { BasicTracerProvider, InMemorySpanExporter, SimpleSpanProcessor } from '@opentelemetry/sdk-trace-base'; + +const propagator = new W3CTraceContextPropagator(); +const exporter = new InMemorySpanExporter(); +const provider = new BasicTracerProvider({spanProcessors: [new SimpleSpanProcessor(exporter)]}); +const tracer = provider.getTracer('contract-probe'); +const vectors = JSON.parse(readFileSync(0, 'utf8')); +const results = await Promise.all(vectors.map(async ({name, carrier, valid}) => { + // Explicit contexts model persisted carriers; ambient polling context must not become a parent. + const extracted = propagator.extract(ROOT_CONTEXT, carrier, defaultTextMapGetter); + const context = trace.getSpanContext(extracted); + assert.equal(Boolean(context), valid, name); + const linked = tracer.startSpan(name, {links: context ? [{context}] : []}, ROOT_CONTEXT); + await Promise.resolve(); + assert.notEqual(linked.spanContext().traceId, context?.traceId); + const outgoing = {}; + propagator.inject(trace.setSpan(ROOT_CONTEXT, linked), outgoing, defaultTextMapSetter); + const roundtrip = {}; + propagator.inject(extracted, roundtrip, defaultTextMapSetter); + linked.end(); + return {name, roundtrip, outgoing}; +})); +await provider.forceFlush(); +const spans = exporter.getFinishedSpans(); +for (const vector of vectors) { + const span = spans.find(s => s.name === vector.name); + assert.equal(span.parentSpanContext, undefined); + assert.equal(span.links.length, vector.valid ? 1 : 0); +} +assert.equal(new Set(results.map(r => r.outgoing.traceparent)).size, vectors.length); +await provider.shutdown(); +process.stdout.write(JSON.stringify(results)); diff --git a/tasks/observability-foundation/experiments/propagation.py b/tasks/observability-foundation/experiments/propagation.py new file mode 100644 index 0000000..dd6769d --- /dev/null +++ b/tasks/observability-foundation/experiments/propagation.py @@ -0,0 +1,116 @@ +# /// script +# requires-python = ">=3.12" +# dependencies = ["opentelemetry-sdk==1.45.0"] +# /// +"""Run with pdm run propagation.py; no application or dependency-file changes.""" + +import asyncio +import io +import json +import logging +from pathlib import Path +import subprocess + +from opentelemetry import trace +from opentelemetry.context import Context +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import SimpleSpanProcessor +from opentelemetry.sdk.trace.export.in_memory_span_exporter import InMemorySpanExporter +from opentelemetry.trace.propagation.tracecontext import TraceContextTextMapPropagator + + +async def main(): + diagnostic = io.StringIO() + diagnostic_handler = logging.StreamHandler(diagnostic) + sdk_logger = logging.getLogger("opentelemetry.trace.span") + sdk_logger.addHandler(diagnostic_handler) + exporter = InMemorySpanExporter() + provider = TracerProvider() + provider.add_span_processor(SimpleSpanProcessor(exporter)) + tracer = provider.get_tracer("contract-probe") + propagator = TraceContextTextMapPropagator() + with tracer.start_as_current_span("python-submit", context=Context()): + carrier = {} + propagator.inject(carrier) + carrier["tracestate"] = "inkcre=synthetic" + parent = carrier["traceparent"] + vectors = [ + {"name": "sampled", "carrier": carrier, "valid": True}, + {"name": "unsampled", "carrier": {"traceparent": parent[:-2] + "00"}, "valid": True}, + { + "name": "future-version", + "carrier": {"traceparent": "01" + parent[2:] + "-extra"}, + "valid": True, + }, + {"name": "missing", "carrier": {}, "valid": False}, + {"name": "malformed", "carrier": {"traceparent": "broken"}, "valid": False}, + { + "name": "zero-id", + "carrier": {"traceparent": "00-" + "0" * 32 + "-" + "1" * 16 + "-01"}, + "valid": False, + }, + { + "name": "bad-state", + "carrier": {"traceparent": parent, "tracestate": "o11y-sensitive-canary"}, + "valid": True, + }, + ] + js = json.loads( + subprocess.check_output( # noqa: S603 + ["node", str(Path(__file__).with_suffix(".mjs"))], + input=json.dumps(vectors), + text=True, + ) + ) + for vector, result in zip(vectors, js, strict=True): + extracted = trace.get_current_span( + propagator.extract(vector["carrier"], Context()) + ).get_span_context() + assert extracted.is_valid == vector["valid"], vector["name"] + restored = trace.get_current_span( + propagator.extract(result["roundtrip"], Context()) + ).get_span_context() + assert (restored.trace_id, restored.span_id, restored.trace_flags) == ( + extracted.trace_id, + extracted.span_id, + extracted.trace_flags, + ) + assert dict(restored.trace_state) == dict(extracted.trace_state) + + async def receive(result): + linked_context = trace.get_current_span( + propagator.extract(result["outgoing"], Context()) + ).get_span_context() + assert linked_context.is_valid + with tracer.start_as_current_span( + result["name"], context=Context(), links=[trace.Link(linked_context)] + ) as span: + await asyncio.sleep(0) + assert trace.get_current_span() is span + assert span.get_span_context().trace_id != linked_context.trace_id + return span.get_span_context().trace_id + + ids = await asyncio.gather(*(receive(result) for result in js)) + assert len(set(ids)) == len(js) + spans = exporter.get_finished_spans()[1:] + assert all(span.parent is None and len(span.links) == 1 for span in spans) + provider.shutdown() + sdk_logger.removeHandler(diagnostic_handler) + result = { + "vectors": [v["name"] for v in vectors], + "python_to_js_to_python": "passed", + "independent_execution_links": "passed", + "python_async_context_isolation": "passed", + "js_explicit_context_isolation": "passed", + "sdk_invalid_state_diagnostic_contains_raw_value": "o11y-sensitive-canary" + in diagnostic.getvalue(), + "browser_runtime": "not tested", + } + (Path(__file__).parent / "runtime" / "propagation.json").write_text( + json.dumps(result, indent=2) + "\n" + ) + print(json.dumps(result, indent=2)) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/tasks/observability-foundation/experiments/readback.py b/tasks/observability-foundation/experiments/readback.py new file mode 100644 index 0000000..858cbdf --- /dev/null +++ b/tasks/observability-foundation/experiments/readback.py @@ -0,0 +1,115 @@ +"""Independently query OpenObserve; exporter success is not this check's oracle.""" + +import base64 +import json +from pathlib import Path +import statistics +import time +import urllib.request + +from lab import credentials + +HERE = Path(__file__).resolve().parent + + +def query(sql, kind): + cred = credentials() + auth = base64.b64encode(f"{cred['email']}:{cred['password']}".encode()).decode() + body = { + "query": { + "sql": sql, + "start_time": int((time.time() - 86400) * 1e6), + "end_time": int((time.time() + 60) * 1e6), + "size": 100, + } + } + request = urllib.request.Request( + "http://127.0.0.1:35080/api/default/_search?type=" + kind, + data=json.dumps(body).encode(), + headers={"Authorization": "Basic " + auth, "Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=20) as response: + result = json.load(response) + assert not result.get("is_partial"), result + return result + + +def main(): + expected = json.loads((HERE / "runtime/expected.json").read_text()) + run_id = expected["run_id"] + assert len(run_id) == 32 and all(c in "0123456789abcdef" for c in run_id) + sql = f"SELECT * FROM default WHERE inkcre_lab_run_id = '{run_id}'" + traces = query(sql, "traces") + logs = query(sql, "logs") + spans = {row["operation_name"]: row for row in traces["hits"]} + assert len(spans) == 3 + executed = spans["job.execute"] + assert executed["trace_id"] == expected["execution_trace_id"] + assert executed.get("reference_parent_span_id") in (None, "") + assert ( + spans["chat synthetic-model"]["reference_parent_span_id"] + == expected["execution_span_id"] + ) + link = json.loads(executed["links"])[0]["context"] + assert link["traceId"] == expected["submission_trace_id"] + assert link["spanId"] == expected["submission_span_id"] + assert executed["trace_id"] != link["traceId"] + assert ( + executed["inkcre_job_id"] == "42" + ) # Observed backend projection, not OTLP's original type. + assert len(logs["hits"]) == 1 + assert logs["hits"][0]["trace_id"] == expected["execution_trace_id"] + assert logs["hits"][0]["span_id"] == expected["execution_span_id"] + assert logs["hits"][0]["inkcre_job_id"] == 42 + ai = spans["chat synthetic-model"] + assert ai["gen_ai_usage_input_tokens"] == 8 + assert ai["gen_ai_usage_output_tokens"] == 3 + metric_data = {} + # Cumulative points are snapshots. Summing repeated exports would double-count them. + for name, value in { + "inkcre_lab_jobs": 3, + "inkcre_lab_duration_count": 2, + "inkcre_lab_duration_sum": 0.4, + }.items(): + result = query(f"SELECT * FROM {name} ORDER BY _timestamp DESC LIMIT 1", "metrics") + assert result["hits"][0]["value"] == value, result + metric_data[name] = result + buckets = query("SELECT * FROM inkcre_lab_duration_bucket", "metrics") + assert {r["le"]: r["value"] for r in buckets["hits"]} == { + "0.1": 1, + "0.25": 1, + "0.5": 2, + "inf": 2, + } + latency = [] + for _ in range(20): + started = time.perf_counter() + assert len(query(sql, "traces")["hits"]) == 3 + latency.append((time.perf_counter() - started) * 1000) + evidence = { + "expected": expected, + "traces": traces, + "logs": logs, + "metrics": metric_data, + "buckets": buckets, + "query_roundtrip_ms": latency, + } + (HERE / "runtime/readback.json").write_text(json.dumps(evidence, indent=2) + "\n") + print( + json.dumps( + { + "three_signals_and_links": "passed", + "small_warm_query_n": 20, + "small_warm_query_p95_ms": sorted(latency)[18], + "small_warm_query_median_ms": statistics.median(latency), + "backend_added_cost_without_input": ai.get("gen_ai_usage_cost"), + "backend_added_agent_version_without_input": ai.get("gen_ai_agent_version"), + "trace_job_attribute_type": type(executed["inkcre_job_id"]).__name__, + }, + indent=2, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/tasks/observability-foundation/experiments/relay-lab.py b/tasks/observability-foundation/experiments/relay-lab.py new file mode 100644 index 0000000..1b553a6 --- /dev/null +++ b/tasks/observability-foundation/experiments/relay-lab.py @@ -0,0 +1,103 @@ +"""Hold the actual core relay and a synthetic OTLP receiver for browser acceptance.""" + +import importlib.util +import json +from pathlib import Path +import signal +import subprocess +import sys +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +import httpx + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +spec = importlib.util.spec_from_file_location("db_lab", HERE / "implementation-db.py") +lab = importlib.util.module_from_spec(spec) +spec.loader.exec_module(lab) +records = [] + + +class Receiver(BaseHTTPRequestHandler): + def do_POST(self): + records.append( + { + "path": self.path, + "body_hex": self.rfile.read(int(self.headers["Content-Length"])).hex(), + "server_auth_ok": self.headers.get("Authorization") + == "Bearer synthetic-relay-ingest", + } + ) + self.send_response(200) + self.send_header("Content-Type", "application/x-protobuf") + self.send_header("Content-Length", "0") + self.end_headers() + + def log_message(self, *args): + pass + + +server = ThreadingHTTPServer(("127.0.0.1", 0), Receiver) +worker = threading.Thread(target=server.serve_forever, daemon=True) +worker.start() +env = { + **lab.environment(), + "SKIP_EXTENSION_START": "1", + "OBSRV__TELEMETRY_ENABLED": "true", + "OTEL_EXPORTER_OTLP_HEADERS": "Authorization=Bearer%20synthetic-relay-ingest", + "PEER_ID": "00000000-0000-4000-8000-000000000002", +} +for signal_name in ("TRACES", "LOGS", "METRICS"): + env[f"OTEL_EXPORTER_OTLP_{signal_name}_ENDPOINT"] = ( + f"http://127.0.0.1:{server.server_port}/v1/{signal_name.lower()}" + ) + env[f"OTEL_EXPORTER_OTLP_{signal_name}_HEADERS"] = "" +with (HERE / "runtime/relay-app.log").open("w") as log: + process = subprocess.Popen( + [ + sys.executable, + "-m", + "uvicorn", + "run:api_app", + "--host", + "127.0.0.1", + "--port", + "35501", + ], + cwd=ROOT, + env=env, + stdout=log, + stderr=subprocess.STDOUT, + ) + try: + for _ in range(300): + if process.poll() is not None: + raise RuntimeError("Core failed; inspect local runtime log") + try: + if httpx.get("http://127.0.0.1:35501/readyz").status_code == 200: + break + except httpx.HTTPError: + pass + time.sleep(0.1) + else: + raise RuntimeError("Core readiness deadline exceeded") + print( + "READY: real core relay http://127.0.0.1:35501/telemetry; synthetic upstream only", + flush=True, + ) + while True: + time.sleep(1) + except KeyboardInterrupt: + pass + finally: + process.send_signal(signal.SIGINT) + process.wait(timeout=40) + server.shutdown() + server.server_close() + worker.join(timeout=2) + (HERE / "runtime/browser-real-relay.json").write_text( + json.dumps(records, indent=2) + "\n" + ) + print(f"Stopped; captured {len(records)} upstream batches without credential storage") diff --git a/tasks/observability-foundation/experiments/runtime-probe.py b/tasks/observability-foundation/experiments/runtime-probe.py new file mode 100644 index 0000000..fffcc55 --- /dev/null +++ b/tasks/observability-foundation/experiments/runtime-probe.py @@ -0,0 +1,460 @@ +"""Disposable PostgreSQL and real HTTP verification of production runtime boundaries.""" + +import asyncio +from contextlib import redirect_stderr +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import io +import json +import os +from pathlib import Path +import runpy +import socket +import sys +import threading +import time +import uuid + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[2] +sys.path.insert(0, str(ROOT)) +LAB = runpy.run_path(str(HERE / "implementation-db.py")) +os.environ.update(LAB["environment"]()) +assert "@127.0.0.1:35432/o11y_impl" in os.environ["DATABASE_URL"] +os.environ["OBSRV__LOGGING_BACKEND"] = "postgresql" +os.environ["OBSRV__TELEMETRY_ENABLED"] = "false" + +import fastapi +from opentelemetry.proto.collector.metrics.v1.metrics_service_pb2 import ( + ExportMetricsServiceRequest, +) +from opentelemetry.proto.collector.trace.v1.trace_service_pb2 import ( + ExportTraceServiceRequest, +) +import psycopg +import pydantic +import uvicorn + +from libs.obsrv.main import setup_obsrv, start_obsrv, close_obsrv +from libs.obsrv.log_record import TRACE_ID +from libs.obsrv.telemetry import close_telemetry, is_enabled, start_telemetry + +LOGGER = setup_obsrv() +from app.business.job import JobHandler, JobManager +from app.business.agent import ( + BoundAgentTool, + InMemoryThreadPersistenceBackend, + Thread, + ThreadState, +) +from app.business.agent.contracts import ToolExecutionError +from app.schemas.ai import FunctionTool, ToolCall +from app.business.peer.http import PeerHTTPOutbound +from app.business.peer.contracts import PeerRequestNotExecuted +from app.engine import ASYNC_DB_ENGINE +from app.middleware import require_peer_jwt +from app.observability import TelemetryMiddleware +from app.scheduler import start_scheduler, drain_scheduler, scheduler, with_trace_id +from app.schemas.job import JobStatus +from app.schemas.peer import PeerModel, PEER_EXECUTION_HEADER + +METRICS_ONLY = "--metrics-only" in sys.argv +RUN = uuid.uuid4().hex +CANARY = "runtime-private-content-" + RUN +JOB_TYPE = "lab.observability.runtime." + RUN +records = [] +seen_legacy_ids = [] +handler_started = asyncio.Event() + + +class Receiver(BaseHTTPRequestHandler): + def do_POST(self): + records.append((self.path, self.rfile.read(int(self.headers["Content-Length"])))) + self.send_response(200) + self.send_header("Content-Length", "0") + self.end_headers() + + def log_message(self, *args): + pass + + +class Parameters(pydantic.BaseModel): + mode: str = "success" + + +class Handler( + JobHandler[Parameters], + job_type=JOB_TYPE, + description="Disposable telemetry boundary probe", + parameters_model=Parameters, + default_timeout_seconds=10, +): + @classmethod + async def can_handle(cls, parameters): + return True + + @classmethod + async def handle(cls, job, parameters): + seen_legacy_ids.append(TRACE_ID.get()) + LOGGER.info(CANARY) + if parameters.mode == "failure": + raise RuntimeError(CANARY) + if parameters.mode == "wait": + handler_started.set() + await asyncio.Event().wait() + job.state = {"finished": True} + + +def spans(): + return [ + span + for path, body in records + if path.endswith("traces") + for resource in ExportTraceServiceRequest.FromString(body).resource_spans + for scope in resource.scope_spans + for span in scope.spans + ] + + +def attributes(span): + return { + attr.key: getattr(attr.value, attr.value.WhichOneof("value")) + for attr in span.attributes + } + + +async def execute(job): + assert job.id is not None + assert await with_trace_id(f"job.{job.id}", JobManager.run)(job.id) + saved = await JobManager.get(job.id) + assert saved is not None + return saved + + +async def main(): + receiver = ThreadingHTTPServer(("127.0.0.1", 0), Receiver) + threading.Thread(target=receiver.serve_forever, daemon=True).start() + endpoint = f"http://127.0.0.1:{receiver.server_port}/v1/" + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + "" if METRICS_ONLY and signal != "METRICS" else endpoint + signal.lower() + ) + diagnostics = io.StringIO() + start_obsrv() + start_scheduler() + await JobManager.sync_job_types() + + # Endpoints alone do not enable telemetry; the existing PG writer is still active. + start_telemetry(enabled=False, service_version="runtime-probe") + off_job = await JobManager.create(JOB_TYPE, {"mode": "success"}) + off_job = await execute(off_job) + assert off_job.status == JobStatus.FINISHED and off_job.submission_traceparent is None + assert not records and not is_enabled() + off_submit_on_execute = await JobManager.create(JOB_TYPE, {"mode": "success"}) + + start_telemetry(enabled=True, service_version="runtime-probe") + linked = await JobManager.create(JOB_TYPE, {"mode": "success"}) + linked = await execute(linked) + assert linked.status == JobStatus.FINISHED + assert bool(linked.submission_traceparent) is not METRICS_ONLY + no_link = await execute(off_submit_on_execute) + assert no_link.submission_traceparent is None + failed = await JobManager.create(JOB_TYPE, {"mode": "failure"}) + failed = await execute(failed) + assert failed.status == JobStatus.FAILED and failed.state["error"] == CANARY + cancelled = await JobManager.create(JOB_TYPE, {"mode": "wait"}) + task = asyncio.create_task( + with_trace_id(f"job.{cancelled.id}", JobManager.run)(cancelled.id) + ) + await asyncio.wait_for(handler_started.wait(), timeout=5) + await JobManager.abort(cancelled.id) + await JobManager.check_abort_requests() + assert await task + cancelled = await JobManager.get(cancelled.id) + assert cancelled.status == JobStatus.ABORTED + on_submit_off_execute = await JobManager.create(JOB_TYPE, {"mode": "success"}) + persisted_parent = on_submit_off_execute.submission_traceparent + assert bool(persisted_parent) is not METRICS_ONLY + + # Concurrent production Tool scopes retain independent success/error outcomes. + entered = 0 + overlapped = asyncio.Event() + + async def tool_handler(parameters): + nonlocal entered + entered += 1 + if entered == 2: + overlapped.set() + await asyncio.wait_for(overlapped.wait(), 2) + if parameters.mode == "failure": + raise ToolExecutionError({"detail": CANARY}) + return {"finished": True} + + backend = InMemoryThreadPersistenceBackend() + tool = BoundAgentTool( + FunctionTool( + id="lab.boundary", + description="Synthetic outcome probe", + input_schema=Parameters.model_json_schema(), + ), + Parameters, + tool_handler, + ) + identifier, state = await backend.create( + ThreadState( + model=1, + tools=(tool.definition,), + tool_choice="auto", + max_model_calls_per_turn=1, + messages=(), + ) + ) + thread = Thread(identifier, state, backend, (tool,)) + tool_results = await asyncio.gather( + *( + thread._execute_tool_call( + ToolCall(id=mode, tool=tool.definition.id, arguments={"mode": mode}) + ) + for mode in ("success", "failure") + ) + ) + assert overlapped.is_set() and [result.is_error for result in tool_results] == [ + False, + True, + ] + + # Production Peer outbound and ASGI telemetry middleware over actual loopback TCP. + app = fastapi.FastAPI() + app.add_middleware(TelemetryMiddleware) + requests = [] + + @app.post("/peer/{item}", dependencies=[fastapi.Depends(require_peer_jwt)]) + async def peer_route(item: str, request: fastapi.Request): + requests.append({"item": item, "traceparent": request.headers.get("traceparent")}) + await request.json() + if item == "not-executed": + return fastapi.responses.JSONResponse( + {"detail": CANARY}, + status_code=503, + headers={PEER_EXECUTION_HEADER: "not-executed"}, + ) + return fastapi.responses.JSONResponse( + {"result": CANARY}, status_code=500 if item == "failed" else 200 + ) + + sock = socket.socket() + sock.bind(("127.0.0.1", 0)) + port = sock.getsockname()[1] + config = uvicorn.Config(app, log_level="critical", access_log=False, lifespan="off") + server = uvicorn.Server(config) + serving = asyncio.create_task(server.serve(sockets=[sock])) + async with asyncio.timeout(5): + while not server.started: + await asyncio.sleep(0.01) + peer = PeerModel(id=uuid.uuid4(), name="runtime-probe") + + async def call_peer(item): + outbound = PeerHTTPOutbound( + peer, {"method": "POST", "url": f"http://127.0.0.1:{port}/peer/{item}"} + ) + payload = { + "body": {"private": CANARY}, + "query": {"private": [CANARY]}, + "headers": {"x-private": [CANARY], "traceparent": [CANARY], "tracestate": [CANARY]}, + } + if item == "not-executed": + try: + await outbound.execute(payload) + raise AssertionError("not-executed must retain its production error") + except PeerRequestNotExecuted: + pass + else: + result = await outbound.execute(payload) + assert result["status"] == (500 if item == "failed" else 200) + + await asyncio.gather(*(call_peer(item) for item in (CANARY, "failed", "not-executed"))) + assert len(requests) == 3 + assert all( + r["traceparent"] == CANARY if METRICS_ONLY else r["traceparent"].startswith("00-") + for r in requests + ) + server.should_exit = True + await serving + sock.close() + await close_telemetry() + + # A new off execution retains the previously captured carrier, including at close. + before = len(records) + preserved = await execute(on_submit_off_execute) + assert preserved.status == JobStatus.FINISHED + assert preserved.submission_traceparent == persisted_parent and len(records) == before + + # Receiver failure cannot alter the durable Job result or PostgreSQL logging. + unused = socket.socket() + unused.bind(("127.0.0.1", 0)) + refused_port = unused.getsockname()[1] + unused.close() + for signal in ("TRACES", "LOGS", "METRICS"): + os.environ[f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT"] = ( + "" + if METRICS_ONLY and signal != "METRICS" + else f"http://127.0.0.1:{refused_port}/v1/{signal.lower()}" + ) + with redirect_stderr(diagnostics): + start_telemetry(enabled=True, service_version="runtime-probe") + outage = await JobManager.create(JOB_TYPE, {"mode": "success"}) + outage = await execute(outage) + start = time.monotonic() + await close_telemetry() + outage_close = time.monotonic() - start + assert outage.status == JobStatus.FINISHED and outage_close < 5 + assert CANARY not in diagnostics.getvalue() + await close_obsrv() + await drain_scheduler() + scheduler.shutdown(wait=False) + await ASYNC_DB_ENGINE.dispose() + receiver.shutdown() + receiver.server_close() + + exported = spans() + assert all(CANARY.encode() not in body for _, body in records) + server_spans = [] + client_spans = [] + if METRICS_ONLY: + assert not exported + else: + by_job = {} + for span in exported: + if span.name in ("job.submit", "job.execute"): + by_job.setdefault(attributes(span)["inkcre.job.id"], {})[span.name] = span + submitted = by_job[linked.id]["job.submit"] + executed = by_job[linked.id]["job.execute"] + assert executed.trace_id != submitted.trace_id and not executed.parent_span_id + assert len(executed.links) == 1 + assert executed.links[0].trace_id == submitted.trace_id + assert executed.links[0].span_id == submitted.span_id + assert not by_job[no_link.id]["job.execute"].links + assert "job.execute" not in by_job[preserved.id] + for span in exported: + assert not span.events and not span.status.message + assert all(CANARY.encode() not in body for _, body in records) + server_spans = [span for span in exported if span.name == "http.server"] + client_spans = [span for span in exported if span.name == "peer.http"] + assert len(server_spans) == len(client_spans) == 3 + assert all(attributes(span)["http.route"] == "/peer/{item}" for span in server_spans) + assert all( + any( + server_span.parent_span_id == client.span_id + and server_span.trace_id == client.trace_id + for client in client_spans + ) + for server_span in server_spans + ) + job_ids = [ + off_job.id, + linked.id, + no_link.id, + failed.id, + cancelled.id, + preserved.id, + outage.id, + ] + assert seen_legacy_ids == [f"job.{job_id}" for job_id in job_ids] + with psycopg.connect(LAB["ADMIN"]) as conn: + pg_logs = conn.execute( + "SELECT trace_id, body FROM inkcre.logs WHERE body=%s", (CANARY,) + ).fetchall() + assert {row[0] for row in pg_logs} == {f"job.{job_id}" for job_id in job_ids} + capacity_rejected = [] + for column in ("submission_traceparent", "submission_tracestate"): + try: + with psycopg.connect(LAB["ADMIN"]) as conn: + conn.execute( + psycopg.sql.SQL("UPDATE inkcre.jobs SET {}=%s WHERE id=%s").format( + psycopg.sql.Identifier(column) + ), + ("x" * 513, off_job.id), + ) + raise AssertionError("database accepted over-capacity carrier") + except psycopg.errors.CheckViolation: + capacity_rejected.append(column) + counts = {} + duration_counts = {} + for path, body in records: + if not path.endswith("metrics"): + continue + batch = ExportMetricsServiceRequest.FromString(body) + for resource in batch.resource_metrics: + for scope in resource.scope_metrics: + for metric in scope.metrics: + if metric.name not in ("inkcre.operation.count", "inkcre.operation.duration"): + continue + points = ( + metric.sum.data_points + if metric.HasField("sum") + else metric.histogram.data_points + ) + for point in points: + labels = {attr.key: attr.value.string_value for attr in point.attributes} + assert set(labels) == {"operation", "outcome"} + key = labels["operation"] + ":" + labels["outcome"] + target = counts if metric.HasField("sum") else duration_counts + value = point.as_int if metric.HasField("sum") else point.count + target[key] = max(value, target.get(key, 0)) + expected = { + "job.submit:success": 4, + "job.execute:success": 2, + "job.execute:error": 1, + "job.execute:cancelled": 1, + "http.server:success": 1, + "http.server:error": 2, + "peer.http:success": 1, + "peer.http:error": 2, + "agent.tool:success": 1, + "agent.tool:error": 1, + } + assert counts == expected, counts + assert duration_counts == expected, duration_counts + evidence = { + "run_id": RUN, + "signal_mode": "metrics-only" if METRICS_ONLY else "traces-logs-metrics", + "operation_counts": counts, + "duration_sample_counts": duration_counts, + "concurrent_tool_outcomes_isolated": True, + "scope": ( + "disposable PostgreSQL + production Job/PG log/Peer HTTP/ASGI mechanisms; " + "not full runtime or SaaS" + ), + "job_ids": job_ids, + "job_statuses": [ + str(job.status) + for job in (off_job, linked, no_link, failed, cancelled, preserved, outage) + ], + "off_with_endpoints_no_export": True, + "independent_execution_trace_with_submission_link": not METRICS_ONLY, + "off_submit_on_execute_no_link": True, + "on_submit_off_execute_carrier_preserved": True, + "postgresql_log_rows": len(pg_logs), + "legacy_job_trace_ids_preserved": True, + "carrier_capacity_rejected": capacity_rejected, + "http_requests": len(requests), + "http_client_spans": len(client_spans), + "http_server_spans": len(server_spans), + "http_parent_child_propagation": not METRICS_ONLY, + "http_errors_not_retried": True, + "canary_absent_at_first_otlp_export": True, + "outage_job_finished": True, + "outage_shutdown_seconds": outage_close, + "trace_spans": len(exported), + "otlp_requests": len(records), + } + destination = ( + HERE + / "evidence" + / ("runtime-probe-metrics-only.json" if METRICS_ONLY else "runtime-probe.json") + ) + destination.write_text(json.dumps(evidence, indent=2) + "\n") + print(json.dumps(evidence, indent=2)) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/tasks/observability-foundation/experiments/sentinel-probe.py b/tasks/observability-foundation/experiments/sentinel-probe.py new file mode 100644 index 0000000..d78d3a2 --- /dev/null +++ b/tasks/observability-foundation/experiments/sentinel-probe.py @@ -0,0 +1,152 @@ +"""Probe -1 sentinels and explicit source-status metadata in OpenObserve 1.0.4.""" + +import json +from pathlib import Path +import time +import urllib.request +import uuid +from readback import query + + +def attrs(values): + return [ + { + "key": k, + "value": {"intValue": str(v)} + if type(v) is int + else {"doubleValue": v} + if type(v) is float + else {"stringValue": v}, + } + for k, v in values.items() + ] + + +def main(): + run_id = uuid.uuid4().hex + now = time.time_ns() + cases = { + "sentinel": { + "gen_ai.usage.input_tokens": -1, + "gen_ai.usage.output_tokens": -1, + "gen_ai.usage.cost": -1.0, + }, + "partial": { + "gen_ai.usage.input_tokens": 8, + "gen_ai.usage.output_tokens": -1, + "gen_ai.usage.cost": -1.0, + }, + "zero": { + "gen_ai.usage.input_tokens": 0, + "gen_ai.usage.output_tokens": 0, + "gen_ai.usage.cost": 0.0, + }, + "known": { + "gen_ai.usage.input_tokens": 8, + "gen_ai.usage.output_tokens": 3, + "gen_ai.usage.cost": 0.125, + }, + "usage_only_sentinel": { + "gen_ai.usage.input_tokens": -1, + "gen_ai.usage.output_tokens": -1, + }, + "source_status": { + "inkcre.ai.usage.input.source": "unavailable", + "inkcre.ai.usage.output.source": "unavailable", + "inkcre.ai.cost.source": "unavailable", + }, + } + spans = [ + { + "name": name, + "traceId": uuid.uuid4().hex, + "spanId": uuid.uuid4().hex[:16], + "startTimeUnixNano": str(now), + "endTimeUnixNano": str(now + 1000), + "attributes": attrs( + { + "inkcre.lab.run_id": run_id, + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": "synthetic", + "gen_ai.request.model": "gpt-4o", + **values, + } + ), + } + for name, values in cases.items() + ] + payload = { + "resourceSpans": [ + { + "resource": {"attributes": attrs({"service.name": "inkcre-sentinel-probe"})}, + "scopeSpans": [{"spans": spans}], + } + ] + } + dest = Path(__file__).parent / "evidence/sentinel-saas-20261003" + dest.mkdir(exist_ok=True) + (dest / "sentinel-input.json").write_text(json.dumps(payload, indent=2) + "\n") + request = urllib.request.Request( + "http://127.0.0.1:34318/v1/traces", + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=10) as response: + assert response.status == 200 + sql = f"SELECT * FROM default WHERE inkcre_lab_run_id = '{run_id}'" + deadline = time.monotonic() + 30 + while True: + result = query(sql, "traces") + if len({row["span_id"] for row in result["hits"]}) == len(cases): + break + assert time.monotonic() < deadline, "OTLP records did not become queryable" + time.sleep(0.5) + rows = {r["operation_name"]: r for r in result["hits"]} + fields = [ + "gen_ai_usage_input_tokens", + "gen_ai_usage_output_tokens", + "gen_ai_usage_total_tokens", + "gen_ai_usage_cost", + "inkcre_ai_usage_input_source", + "inkcre_ai_cost_source", + ] + summary = {name: {k: row.get(k) for k in fields} for name, row in rows.items()} + assert all( + rows["sentinel"][k] == -1 + for k in [ + "gen_ai_usage_input_tokens", + "gen_ai_usage_output_tokens", + "gen_ai_usage_cost", + ] + ) + assert rows["source_status"]["inkcre_ai_usage_input_source"] == "unavailable" + assert ( + rows["known"]["gen_ai_usage_cost"] == 0.125 + and rows["zero"]["gen_ai_usage_input_tokens"] == 0 + ) + # OTLP retry can deliver the same span twice; isolate sentinel arithmetic from duplicates. + unique_rows = f"SELECT DISTINCT span_id, gen_ai_usage_input_tokens, gen_ai_usage_cost FROM default WHERE inkcre_lab_run_id = '{run_id}' AND operation_name IN ('sentinel','zero','known')" # noqa: E501 + aggregates = query( + f"SELECT sum(gen_ai_usage_input_tokens) AS naive_input, sum(gen_ai_usage_cost) AS naive_cost, sum(CASE WHEN gen_ai_usage_input_tokens >= 0 THEN gen_ai_usage_input_tokens ELSE NULL END) AS known_input, sum(CASE WHEN gen_ai_usage_cost >= 0 THEN gen_ai_usage_cost ELSE NULL END) AS known_cost FROM ({unique_rows})", # noqa: E501 + "traces", + ) + assert aggregates["hits"][0] == { + "naive_input": 7, + "naive_cost": -0.875, + "known_input": 8, + "known_cost": 0.125, + } + report = { + "input": payload, + "readback": result, + "duplicate_rows": len(result["hits"]) - len(cases), + "summary": summary, + "aggregate_readback": aggregates, + "scope": "Synthetic fixed-version backend behavior only; not a production mapping or Cloud test", # noqa: E501 + } + (dest / "sentinel.json").write_text(json.dumps(report, indent=2) + "\n") + print(json.dumps({"summary": summary, "aggregates": aggregates["hits"]}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tasks/observability-foundation/experiments/sentinel-saas-20261003.md b/tasks/observability-foundation/experiments/sentinel-saas-20261003.md new file mode 100644 index 0000000..ed25eea --- /dev/null +++ b/tasks/observability-foundation/experiments/sentinel-saas-20261003.md @@ -0,0 +1,67 @@ +# 2026-10-03:缺失值与 SaaS 方向修订 + +Sir 明确把 serverless、scale-to-0 和无需维护常驻服务作为选型标准。本轮重新判断 `-1`,并查阅三家 SaaS 的官方文档;没有开通账户、采购服务或上传任何业务数据。自建五组件的既有实验仍有效,但不再拥有默认部署结论,SaaS 方向见 D8;Sir 随后明确暂时不能付费,D9 明确零费用;随后 Sir 通过 D10 选定 Grafana Cloud Free、默认关闭新增遥测并保留 PG。现行决定见[任务决定](../decisions.md),下文候选顺序保留为当时的判断。 + +## `-1` 的实际效果 + +复用前轮固定 digest 的 OpenObserve 1.0.4 和 Collector 0.162.0,仅发送六条合成 span。输入与独立 API 查询见 [sentinel.json](evidence/sentinel-saas-20261003/sentinel.json),可复现脚本为 [sentinel-probe.py](sentinel-probe.py)。本轮不验证 OpenObserve Cloud 的运行版本或 AI UI。 + +| 输入情况 | 后端原始字段 | 后端自动生成字段 | +| --- | --- | --- | +| input/output/cost 都为 `-1` | 三个 `-1` 保留 | total 为 0 | +| input=8、output/cost=`-1` | 原始值保留 | total 为 8,不能解释成完整总量 | +| 明确零 | input/output/cost 为 0 | total 为 0 | +| input=8、output=3、cost=0.125 | 原始值保留 | total 为 11 | +| 只有 input/output=`-1`,省略 cost | 两个 `-1` 保留 | total 为 0,cost 为 -0.0 | +| 省略数值,同时声明各字段 source=`unavailable` | source 字符串保留 | 数值仍补零,但来源元数据可以识别 unknown | + +把第一、第三、第四条直接求和,input=7、cost=-0.875;先过滤负值后是 8 与 0.125。查询使用 DISTINCT span_id 排除传输重试重复,以单独判别 sentinel 算术。过滤后的值也只是已知部分,必须同时显示 unknown 数量;全部缺失时不能展示为完整的零。input/output 部分缺失时,完整 total 仍 unknown。 + +结论是 `-1` 可作特定存储/展示的缺失编码,但不适合直接进入标准 usage 指标、成本总额或默认 AI 面板。源端标准字段仍按实际 provider 值写入,缺失省略;如果所选后端会补值,允许最少的逐字段来源元数据,例如 input/output 的 provider/unavailable、cost 的 estimate/unavailable。查询先解释来源,再计算;这类元数据没有复制数值,也不另建业务状态权威。后端生成的估算、total 和 agent.version 都不能未经判别就当作采集事实。无需因为该缺口而承担五组件常驻运维。 + +## SaaS 的当前能力与费用 + +以下是 2026-10-03 读取的官方公开资料,美元价格不含税,不是账户报价或实测账单。预算要计入采集端计算/出口、写入、查询、保留和超额;业务 scale-to-0 不保证所有留存、历史查询或平台项目都零费。 + +| 候选 | 官方能力与费用 | 对本任务的判断 | +| --- | --- | --- | +| OpenObserve Cloud | 统一 logs/traces/metrics;当前 $0.50/GB 写入、$0.01/GB 查询,14 天试用;非 metrics 默认 30 天、metrics 15 个月。2025 定价政策说明无最低消费、取消永久免费档 | 第一验证候选;符合用户偏好、统一查询和按量费用,需在 Cloud 验来源字段、必要因果查询及预算边界 | +| Grafana Cloud Free | 托管三信号并接受 SDK 直发 OTLP;免费档 10k 活跃指标序列、每月 logs/traces 各 50GB、14 天保留、3 名活跃用户。Pro 有 $19/月平台费另加用量 | 零美元试点的明确备选;组件运维由供应商负责,不能与自建五组件混同 | +| PostHog Cloud | AI O11y、logs、通用 tracing 与原生 OTLP metrics;tracing 为 Beta、metrics 为开放 Alpha。每月免费 10GB logs、100k AI events;logs 默认 14 天 | 有吸引力的整合候选,但不作为当前唯一基建首选;metrics/通用 tracing 费用、必要 Link 查询和账户行为尚未确认 | + +价格来源:[OpenObserve 当前价格](https://openobserve.ai/pricing/)、[无最低消费与免费档调整](https://openobserve.ai/blog/june-25-pricing-policy-updates/)、[Grafana Cloud 价格](https://grafana.com/pricing/)、[PostHog 价格](https://posthog.com/pricing/)。Grafana 官方允许不能运行 Collector 的应用 [直接用 SDK 发送 OTLP](https://grafana.com/docs/grafana-cloud/send-data/otlp/send-data-otlp/)。免费额度不是容量测量,达到预算上限可能丢弃新数据;PostHog 价格页明确说明了此行为。 + +PostHog 的 [普通 tracing](https://posthog.com/docs/distributed-tracing) 与 [AI OTLP](https://posthog.com/docs/ai-observability/installation/opentelemetry) 是不同出口:`/i/v1/traces` 接通用 span,`/i/v0/ai/otel` 只接符合 AI 名称/属性的 span。不能用 AI 出口代替整个 Job trace。其 [metrics 文档](https://posthog.com/docs/metrics) 提供标准 `/i/v1/metrics`,不是只能从日志派生指标;Alpha 明确提示接入细节可能变更。 + +[官方 AI/Trace 关联说明](https://posthog.com/docs/distributed-tracing/link-ai-observability)说明两类记录的 ID 编码不同,分属不同查询集群,需两次查询,当前 UI 没有互相跳转。提供同一 trace ID 可以关联,并不等于已有一体化诊断体验。未知值、部分 usage、重试重复、cost 来源与 Span Links 都要在目标服务实际读回,不能由品牌或“支持 OTLP”推断。 + +## 准入边界与结束状态 + +默认方案改为应用 SDK 直发 SaaS;浏览器能否直发取决于供应商是否提供适合客户端的受限写入能力。否则复用受控的按请求转发入口,禁止把服务端秘密打入公开客户端。Collector 仅在已证明的协议、认证、缓存或路由缺口需要它时引入。短生命周期在平台保证的运行窗口内有界 flush,失败允许丢失,不能依赖响应完成后继续运行。 + +一次候选验收覆盖三信号、Job→执行→提交关联、AI unknown/zero/partial、浏览器入口、导出样本、限额和应用停止;达到必要诊断能力即停止比选。不要求所有 OTLP 字段原样存储,不以历史 Link flags 或当前不用的派生字段单独否决产品。因果 ID、来源事实和正确聚合不可丢。 + +复现顺序是 `lab.py up`、等待 ready、`sentinel-probe.py`、`lab.py stop`,从 core-py 根用 python3 运行;远端与命令级 WSL interop 沿用[前轮说明](README.md)。已停止本轮两个容器和隧道,合成卷随父任务保留。完整资源与任务包验证结果另归档于本轮 evidence 目录。没有修改应用、Hub、共享引用或 SVC 开发数据库。 + +## Cloudflare 与当前零费用方案 + +Sir 随后明确暂时不能使用收费服务。因此当前新增观测服务费用按 **$0** 处理,付费服务的试用期不能充当长期免费方案。OpenObserve Cloud 从第一候选移出,先验 Grafana Cloud Free;Cloudflare 原生能力用于已在该平台运行的 Unit 或明确需要的模型网关,不强制业务迁移平台。 + +| Cloudflare 能力 | 2026-10-03 核对的官方范围 | 本任务适用性 | +| --- | --- | --- | +| Workers Logs、Workers Traces、Cloudflare Traces | 采集 Worker 与 Cloudflare 网络/服务路径。Workers Logs 当前 Free 为每天20万条、保留3天;最新 Workers Traces 页面写7天保留 | 可覆盖 ext-reg 等实际运行于 Cloudflare 的 Unit;没有找到接收任意外部 Peer 三信号的通用 OTLP 存储入口,不能由“支持 OTel 导出”推导有此入口 | +| AI Gateway | 核心功能免费,观测经过网关的模型调用,并可向外导出 GenAI spans;模型推理本身仍按 provider 规则收费 | 可用于确有模型代理需求的路径,不提供完整 Job/检索/工具/跨 Peer 因果图;不为免费观测强制改变模型请求路径 | +| Workers Analytics Engine | 公布 Free 每天10万写点、1万查询;页面说明当前尚未计费。数据保留3个月,通过 Worker writeDataPoint 写入,读写均可自适应采样 | 适合聚合趋势;不保证找回单条记录或还原事件序列,不选作唯一 Trace/长期证据库 | +| Worker 请求转发 | Free 每天10万请求,每次10ms CPU;无需常驻主机 | 浏览器私密出口的可选转发宿主,需先证明现有入口不能满足;不附带 D1/R2、自研索引或重试队列 | + +来源:[Workers Logs](https://developers.cloudflare.com/workers/observability/logs/workers-logs/)、[Workers Traces](https://developers.cloudflare.com/workers/observability/traces/)、[Cloudflare 数据集](https://developers.cloudflare.com/observability/logs/datasets/)、[Trace 配置](https://developers.cloudflare.com/observability/traces/configuration/)、[AI Gateway 价格](https://developers.cloudflare.com/ai-gateway/reference/pricing/)、[Analytics Engine 价格](https://developers.cloudflare.com/analytics/analytics-engine/pricing/)、[保留](https://developers.cloudflare.com/analytics/analytics-engine/limits/)、[采样限制](https://developers.cloudflare.com/analytics/analytics-engine/sampling/)、[Workers 价格](https://developers.cloudflare.com/workers/platform/pricing/)。这些是官方能力与限额,不是目标账户实测。 + +Cloudflare 的日期边界必须保留:[统一 Observability 定价](https://developers.cloudflare.com/observability/pricing/)从 **2026-12-01** 才生效,Free 对列出的日志/Trace 共享每天0.5GB、保留7天,超限停止新摄取直到 UTC 零点,已存数据继续可查。不能把它写成10月3日已生效,也不能假定所有安全日志数据集均免费。Workers Logs 当前页仍列3天,而新数据集总表列7天;当前日志使用专门页面的明确旧定价,启用时核对账户实际设置,不混用即将切换的政策或 previews 文档。 + +AI Gateway 同样存在新旧客户差异:[9月24日之后首次建网关](https://developers.cloudflare.com/ai-gateway/reference/limits/)的日志跟随 Workers Logs;之前的客户才有 legacy 免费共10万条存储。不能把旧价格表当作新账户承诺。[日志配置](https://developers.cloudflare.com/ai-gateway/observability/logging/)支持 `cf-aig-collect-log-payload: false` 保留 metadata 而不保存原始请求响应。但 [OTel exporter](https://developers.cloudflare.com/ai-gateway/observability/otel-integration/)单独列出了 prompt/completion 属性;前述日志开关是否同时约束导出内容尚未验证,不能直接打开 exporter 并声称原文不会外流。Gateway 处理请求本身也引入模型内容经过 Cloudflare 的业务路径,不是纯旁路诊断。 + +当前工程推荐是 **标准 SDK → Grafana Cloud Free**,适用的 Cloudflare Unit 另以现成 OTLP 导出汇入同一诊断入口。Cloudflare 有[官方 Grafana 导出步骤](https://developers.cloudflare.com/observability/export/opentelemetry/grafana-cloud/),但 Workers 的[平台 OTLP 出口](https://developers.cloudflare.com/workers/observability/opentelemetry-export/)不支持平台/自定义 metrics 导出,不能把 logs/traces 导出视为三信号全覆盖。内容准入、外部传播与指标补充按 Unit 实测,不为完整面板强制新增常驻采集器。 + +Grafana 官方明确 [Free 不需要信用卡且非限时试用](https://grafana.com/docs/grafana-cloud/learn-and-build/get-started/learn/),当前[免费额度](https://grafana.com/pricing/)为10k活跃指标序列、logs/traces各50GB/月、14天保留。只用实际 Free,不开 Pro、超额付费或依赖 trial 附加能力。验收记录目标账号的序列/摄取/速率限制、超限拒收与采集端丢弃;免费额度不是完整诊断或生产 SLA 保证。必要时减少日志量、采样和指标基数,业务结果与长期证据继续由已有 owner 保存。 + +用 Cloudflare Workers 加 D1/R2 自建观测平台虽然能利用基础设施额度,但会把 OTLP 转换、索引、查询、Trace 展示和数据生命周期交给 InKCre;与当前少维护、可替换的目标不符。本轮不采用,也没有开通任何云端功能、修改服务配置或发送遥测。 diff --git a/tasks/observability-foundation/experiments/signals.py b/tasks/observability-foundation/experiments/signals.py new file mode 100644 index 0000000..d3a8628 --- /dev/null +++ b/tasks/observability-foundation/experiments/signals.py @@ -0,0 +1,117 @@ +# /// script +# requires-python = ">=3.12" +# dependencies = ["opentelemetry-sdk==1.45.0", "opentelemetry-exporter-otlp-proto- +# http==1.45.0"] +# /// +"""Send one synthetic causal chain, a correlated log, counter and histogram.""" + +import json +from pathlib import Path +import uuid + +from opentelemetry import trace +from opentelemetry._logs import SeverityNumber +from opentelemetry.context import Context +from opentelemetry.exporter.otlp.proto.http._log_exporter import OTLPLogExporter +from opentelemetry.exporter.otlp.proto.http.metric_exporter import OTLPMetricExporter +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk._logs import LoggerProvider +from opentelemetry.sdk._logs.export import SimpleLogRecordProcessor +from opentelemetry.sdk.metrics import MeterProvider +from opentelemetry.sdk.metrics.export import PeriodicExportingMetricReader +from opentelemetry.sdk.metrics.view import ExplicitBucketHistogramAggregation, View +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import SimpleSpanProcessor + + +def main(): + run_id = uuid.uuid4().hex + endpoint = "http://127.0.0.1:34318/v1/" + resource = Resource.create( + { + "service.name": "inkcre-o11y-synthetic", + "service.version": "g1", + "inkcre.deployment.id": "synthetic-g1", + "inkcre.peer.id": "synthetic-python", + } + ) + traces = TracerProvider(resource=resource) + traces.add_span_processor( + SimpleSpanProcessor(OTLPSpanExporter(endpoint=endpoint + "traces", timeout=3)) + ) + logs = LoggerProvider(resource=resource) + logs.add_log_record_processor( + SimpleLogRecordProcessor(OTLPLogExporter(endpoint=endpoint + "logs", timeout=3)) + ) + metrics = MeterProvider( + resource=resource, + metric_readers=[ + PeriodicExportingMetricReader( + OTLPMetricExporter(endpoint=endpoint + "metrics", timeout=3), + export_interval_millis=3600000, + ) + ], + views=[ + View( + instrument_name="inkcre_lab_duration", + aggregation=ExplicitBucketHistogramAggregation([0.1, 0.25, 0.5]), + ) + ], + ) + tracer = traces.get_tracer("inkcre-g1") + attributes = {"inkcre.job.id": 42, "inkcre.lab.run_id": run_id} + with tracer.start_as_current_span( + "job.submit", context=Context(), attributes=attributes + ) as submit: + submitted = submit.get_span_context() + with tracer.start_as_current_span( + "job.execute", context=Context(), links=[trace.Link(submitted)], attributes=attributes + ) as job: + executed = job.get_span_context() + with tracer.start_as_current_span( + "chat synthetic-model", + attributes={ + **attributes, + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": "synthetic", + "gen_ai.usage.input_tokens": 8, + "gen_ai.usage.output_tokens": 3, + "gen_ai.response.finish_reasons": ["stop"], + }, + ): + pass + logs.get_logger("inkcre-g1").emit( + body="synthetic job finished", + severity_number=SeverityNumber.INFO, + attributes=attributes, + ) + meter = metrics.get_meter("inkcre-g1") + meter.create_counter("inkcre_lab_jobs", unit="{job}").add(3, {"outcome": "finished"}) + duration = meter.create_histogram("inkcre_lab_duration", unit="s") + for value in [0.1, 0.3]: + duration.record(value, {"operation": "synthetic"}) + metrics.force_flush() + traces.force_flush() + logs.force_flush() + metrics.shutdown() + traces.shutdown() + logs.shutdown() + result = { + "run_id": run_id, + "submission_trace_id": format(submitted.trace_id, "032x"), + "submission_span_id": format(submitted.span_id, "016x"), + "execution_trace_id": format(executed.trace_id, "032x"), + "execution_span_id": format(executed.span_id, "016x"), + "expected_job_counter": 3, + "expected_duration_count": 2, + "expected_duration_sum": 0.4, + } + (Path(__file__).parent / "runtime" / "expected.json").write_text( + json.dumps(result, indent=2) + "\n" + ) + print(json.dumps(result, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tasks/observability-foundation/experiments/stack-lab.py b/tasks/observability-foundation/experiments/stack-lab.py new file mode 100644 index 0000000..c043569 --- /dev/null +++ b/tasks/observability-foundation/experiments/stack-lab.py @@ -0,0 +1,370 @@ +"""Five-component synthetic lab, capped at a combined 2 CPU / 4 GiB.""" + +import json +from pathlib import Path +import subprocess +import sys + +from lab import docker, HERE, HOST, IMAGES + +PROJECT = "inkcre-o11y-stack-g1-b0a97f7c" +SOCKET = "/tmp/inkcre-o11y-stack-b0a97f7c.sock" +IMAGES = { + **IMAGES, + "tempo": "grafana/tempo@sha256:3076b8dcdfb32fd6bc5ccef85e7b7313e6199b9cb84366257fc17ecb696db5fd", # noqa: E501 + "loki": "grafana/loki@sha256:1107dd5274e0ada47e42472b7a7e71f3b2a2fe878878108f3e2f9e51528f0193", # noqa: E501 + "prometheus": "prom/prometheus@sha256:efd719c99d83b060d9daefdcf00360461adf279f45ef5391f8d111892118753e", # noqa: E501 + "grafana": "grafana/grafana@sha256:b28bae15e219c998fb0e0424ed724930cc61b1f61fb404d47c862f9a23f9e572", # noqa: E501 +} +PORTS = {34318: 4318, 33200: 3200, 33100: 3100, 39090: 9090, 33001: 3000, 38888: 8888} + + +def compose(): + exporters = {} + for name, endpoint in { + "tempo": "http://tempo:4318", + "loki": "http://loki:3100/otlp", + "prometheus": "http://prometheus:9090/api/v1/otlp", + }.items(): + exporters["otlp_http/" + name] = { + "endpoint": endpoint, + "timeout": "2s", + "retry_on_failure": { + "initial_interval": "1s", + "max_interval": "2s", + "max_elapsed_time": "10s", + }, + "sending_queue": {"sizer": "bytes", "queue_size": 1048576, "num_consumers": 1}, + } + collector = { + "receivers": {"otlp": {"protocols": {"http": {"endpoint": "0.0.0.0:4318"}}}}, + "processors": { + "memory_limiter": {"check_interval": "1s", "limit_mib": 384, "spike_limit_mib": 64}, + "batch": {"timeout": "1s", "send_batch_size": 128, "send_batch_max_size": 128}, + }, + "exporters": exporters, + "service": { + "telemetry": { + "metrics": { + "readers": [ + {"pull": {"exporter": {"prometheus": {"host": "0.0.0.0", "port": 8888}}}} + ] + } + }, + "pipelines": { + signal: { + "receivers": ["otlp"], + "processors": ["memory_limiter", "batch"], + "exporters": ["otlp_http/" + name], + } + for signal, name in [ + ("traces", "tempo"), + ("logs", "loki"), + ("metrics", "prometheus"), + ] + }, + }, + } + loki = { + "auth_enabled": False, + "server": {"http_listen_port": 3100, "log_level": "warn"}, + "common": { + "instance_addr": "127.0.0.1", + "path_prefix": "/loki", + "storage": { + "filesystem": {"chunks_directory": "/loki/chunks", "rules_directory": "/loki/rules"} + }, + "replication_factor": 1, + "ring": {"kvstore": {"store": "inmemory"}}, + }, + "schema_config": { + "configs": [ + { + "from": "2024-01-01", + "store": "tsdb", + "object_store": "filesystem", + "schema": "v13", + "index": {"prefix": "index_", "period": "24h"}, + } + ] + }, + "limits_config": { + "allow_structured_metadata": True, + "retention_period": "168h", + "otlp_config": { + "resource_attributes": { + "ignore_defaults": True, + "attributes_config": [ + { + "action": "index_label", + "attributes": ["service.name", "inkcre.deployment.id"], + } + ], + } + }, + }, + "compactor": { + "working_directory": "/loki/compactor", + "retention_enabled": True, + "delete_request_store": "filesystem", + }, + "analytics": {"reporting_enabled": False}, + "pattern_ingester": {"enabled": False}, + } + prometheus = { + "global": {"scrape_interval": "15s"}, + "otlp": { + "promote_resource_attributes": [ + "service.name", + "service.instance.id", + "inkcre.deployment.id", + ], + "translation_strategy": "UnderscoreEscapingWithSuffixes", + }, + "scrape_configs": [ + {"job_name": "collector", "static_configs": [{"targets": ["collector:8888"]}]} + ], + } + datasources = { + "apiVersion": 1, + "datasources": [ + { + "name": "Tempo", + "type": "tempo", + "uid": "tempo", + "url": "http://tempo:3200", + "access": "proxy", + "jsonData": { + "tracesToLogsV2": { + "datasourceUid": "loki", + "spanStartTimeShift": "-1m", + "spanEndTimeShift": "1m", + "filterByTraceID": True, + "tags": [{"key": "service.name", "value": "service_name"}], + } + }, + }, + { + "name": "Loki", + "type": "loki", + "uid": "loki", + "url": "http://loki:3100", + "access": "proxy", + "jsonData": { + "derivedFields": [ + { + "name": "TraceID", + "matcherType": "label", + "matcherRegex": "trace_id", + "url": "$${__value.raw}", + "datasourceUid": "tempo", + } + ] + }, + }, + { + "name": "Prometheus", + "type": "prometheus", + "uid": "prometheus", + "url": "http://prometheus:9090", + "access": "proxy", + "isDefault": True, + }, + ], + } + dashboard = { + "uid": "inkcre-o11y-g1", + "title": "InKCre G1 合成诊断", + "schemaVersion": 39, + "time": {"from": "now-1h", "to": "now"}, + "templating": { + "list": [ + { + "name": "job", + "type": "textbox", + "label": "Job ID", + "query": "42", + "current": {"text": "42", "value": "42"}, + } + ] + }, + "panels": [ + { + "id": 1, + "title": "Job 日志 → 展开 TraceID", + "type": "logs", + "gridPos": {"x": 0, "y": 0, "w": 24, "h": 10}, + "datasource": {"type": "loki", "uid": "loki"}, + "targets": [ + { + "refId": "A", + "expr": '{service_name="inkcre-o11y-synthetic"} | inkcre_job_id="$job"', + } + ], + "options": {"showTime": True, "wrapLogMessage": True, "enableLogDetails": True}, + }, + { + "id": 2, + "title": "完成计数(服务聚合,不按 Job 打标签)", + "type": "stat", + "gridPos": {"x": 0, "y": 10, "w": 12, "h": 6}, + "datasource": {"type": "prometheus", "uid": "prometheus"}, + "targets": [ + { + "refId": "A", + "expr": "sum(last_over_time(inkcre_lab_jobs_total[1h]))", + "instant": True, + } + ], + }, + { + "id": 3, + "title": "合成耗时观测数", + "type": "stat", + "gridPos": {"x": 12, "y": 10, "w": 12, "h": 6}, + "datasource": {"type": "prometheus", "uid": "prometheus"}, + "targets": [ + { + "refId": "A", + "expr": "sum(last_over_time(inkcre_lab_duration_seconds_count[1h]))", + "instant": True, + } + ], + }, + ], + } + configs = { + "collector": json.dumps(collector), + "loki": json.dumps(loki), + "prometheus": json.dumps(prometheus), + "tempo": (HERE / "tempo.yaml").read_text(), + "datasources": json.dumps(datasources), + "dashboard": json.dumps(dashboard, ensure_ascii=False), + "dashboards": json.dumps( + { + "apiVersion": 1, + "providers": [ + { + "name": "task-lab", + "type": "file", + "options": {"path": "/var/lib/grafana/task-dashboards"}, + } + ], + } + ), + } + services = {} + for name, cpu, memory, port, path, volume in [ + ("tempo", 0.5, 1024, 33200, "/etc/tempo/lab.yaml", "/var/tempo"), + ("loki", 0.5, 1024, 33100, "/etc/loki/lab.yaml", "/loki"), + ("prometheus", 0.35, 768, 39090, "/etc/prometheus/prometheus.yml", "/prometheus"), + ("grafana", 0.4, 768, 33001, None, "/var/lib/grafana"), + ("collector", 0.25, 512, 34318, "/etc/otelcol/lab.yaml", None), + ]: + service = { + "image": IMAGES[name], + "cpus": cpu, + "mem_limit": f"{memory}m", + "ports": [f"127.0.0.1:{port}:{PORTS[port]}"], + "configs": [], + } + if volume: + service["volumes"] = [f"{name}:{volume}"] + if path: + service["configs"].append({"source": name, "target": path}) + if name == "tempo": + service["command"] = ["-target=all", "-config.file=" + path] + if name == "loki": + service["command"] = ["-config.file=" + path] + if name == "prometheus": + service["command"] = [ + "--config.file=" + path, + "--web.enable-otlp-receiver", + "--storage.tsdb.retention.time=30d", + "--storage.tsdb.retention.size=512MB", + ] + if name == "collector": + service["command"] = ["--config=" + path] + service["ports"].append("127.0.0.1:38888:8888") + if name == "grafana": + # Synthetic loopback lab only; Editor is needed for Explore in Grafana OSS. + service["environment"] = { + "GF_AUTH_ANONYMOUS_ENABLED": "true", + "GF_AUTH_ANONYMOUS_ORG_ROLE": "Editor", + "GF_AUTH_DISABLE_LOGIN_FORM": "true", + "GF_ANALYTICS_REPORTING_ENABLED": "false", + "GF_ANALYTICS_CHECK_FOR_UPDATES": "false", + "GF_ANALYTICS_CHECK_FOR_PLUGIN_UPDATES": "false", + "GF_PLUGINS_PREINSTALL_DISABLED": "true", + } + service["configs"] = [ + {"source": key, "target": target} + for key, target in [ + ("datasources", "/etc/grafana/provisioning/datasources/lab.yaml"), + ("dashboards", "/etc/grafana/provisioning/dashboards/lab.yaml"), + ("dashboard", "/var/lib/grafana/task-dashboards/lab.json"), + ] + ] + services[name] = service + return json.dumps( + { + "services": services, + "configs": { + name: {"content": value.replace("$", "$$")} for name, value in configs.items() + }, + "volumes": {name: {} for name in ["tempo", "loki", "prometheus", "grafana"]}, + } + ) + + +def main(action): + if action == "up": + print(docker("compose", "-p", PROJECT, "-f", "-", "up", "-d", data=compose())) + if not Path(SOCKET).exists(): + forwards = [ + part for port in PORTS for part in ["-L", f"127.0.0.1:{port}:127.0.0.1:{port}"] + ] + subprocess.run( # noqa: S603 + [ + "ssh", + "-M", + "-S", + SOCKET, + "-fNT", + "-o", + "BatchMode=yes", + "-o", + "ExitOnForwardFailure=yes", + *forwards, + HOST, + ], + check=True, + ) + elif action in {"stop", "remove"}: + args = ["stop"] if action == "stop" else ["down", "--volumes"] + print(docker("compose", "-p", PROJECT, "-f", "-", *args, data=compose())) + if Path(SOCKET).exists(): + subprocess.run(["ssh", "-S", SOCKET, "-O", "exit", HOST], check=True) # noqa: S603 + elif action == "stats": + print( + docker( + "stats", + "--no-stream", + "--format", + "{{json .}}", + *[ + f"{PROJECT}-{name}-1" + for name in ["tempo", "loki", "prometheus", "grafana", "collector"] + ], + ) + ) + elif action == "logs": + print( + docker("compose", "-p", PROJECT, "-f", "-", "logs", "--tail", "20", data=compose()) + ) + else: + raise SystemExit("Use up, stop, remove, stats, or logs") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tasks/observability-foundation/experiments/stack-probe.py b/tasks/observability-foundation/experiments/stack-probe.py new file mode 100644 index 0000000..e72b3e8 --- /dev/null +++ b/tasks/observability-foundation/experiments/stack-probe.py @@ -0,0 +1,174 @@ +"""Query the same SDK sample and an absent/zero log discriminator from independent APIs.""" + +import base64 +import json +from pathlib import Path +import statistics +import sys +import time +import urllib.parse +import urllib.request + +HERE = Path(__file__).resolve().parent + + +def get(base, path, params=None): + url = base + path + ("?" + urllib.parse.urlencode(params) if params else "") + with urllib.request.urlopen(url, timeout=20) as response: + return json.load(response) + + +def main(action): + expected = json.loads((HERE / "runtime/expected.json").read_text()) + if action == "send-log-cases": + now = time.time_ns() + records = [ + { + "timeUnixNano": str(now + i), + "body": {"stringValue": "synthetic " + name}, + "attributes": [ + {"key": "inkcre.job.id", "value": {"intValue": "42"}}, + {"key": "inkcre.lab.case", "value": {"stringValue": name}}, + *( + [{"key": "gen_ai.usage.input_tokens", "value": {"intValue": "0"}}] + if name == "zero" + else [] + ), + ], + "traceId": expected["execution_trace_id"], + "spanId": expected["execution_span_id"], + } + for i, name in enumerate(["absent", "zero"]) + ] + payload = { + "resourceLogs": [ + { + "resource": { + "attributes": [ + {"key": "service.name", "value": {"stringValue": "inkcre-o11y-synthetic"}} + ] + }, + "scopeLogs": [{"logRecords": records}], + } + ] + } + request = urllib.request.Request( + "http://127.0.0.1:34318/v1/logs", + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=5) as response: + assert response.status == 200 + (HERE / "runtime/stack-log-input.json").write_text(json.dumps(payload, indent=2) + "\n") + print("Synthetic absent/zero log cases sent") + return + if action not in {"direct", "grafana"}: + raise SystemExit("Use send-log-cases, direct, or grafana") + bases = ( + { + "loki": "http://127.0.0.1:33100", + "prometheus": "http://127.0.0.1:39090", + "tempo": "http://127.0.0.1:33200", + } + if action == "direct" + else { + name: "http://127.0.0.1:33001/api/datasources/proxy/uid/" + name + for name in ["loki", "prometheus", "tempo"] + } + ) + log_query = { + "query": '{service_name="inkcre-o11y-synthetic"} | inkcre_job_id="42"', + "start": str(time.time_ns() - 3600 * 10**9), + "limit": 100, + } + logs = get(bases["loki"], "/loki/api/v1/query_range", log_query) + entries = logs["data"]["result"] + sdk_log = next( + row for row in entries if row["stream"].get("inkcre_lab_run_id") == expected["run_id"] + ) + assert sdk_log["stream"]["trace_id"] == expected["execution_trace_id"] + assert sdk_log["stream"]["span_id"] == expected["execution_span_id"] + cases = { + row["stream"]["inkcre_lab_case"]: row + for row in entries + if "inkcre_lab_case" in row["stream"] + } + assert "gen_ai_usage_input_tokens" not in cases["absent"]["stream"] + assert cases["zero"]["stream"]["gen_ai_usage_input_tokens"] == "0" + traces = { + key: get(bases["tempo"], "/api/v2/traces/" + expected[key]) + for key in ["submission_trace_id", "execution_trace_id"] + } + spans = [ + span + for resource in traces["execution_trace_id"]["trace"]["resourceSpans"] + for scope in resource["scopeSpans"] + for span in scope["spans"] + ] + executed = next(span for span in spans if span["name"] == "job.execute") + assert ( + base64.b64decode(executed["links"][0]["traceId"]).hex() + == expected["submission_trace_id"] + ) + job_traces = get( + bases["tempo"], + "/api/search", + { + "q": "{ span.inkcre.job.id = 42 }", + "start": int(time.time() - 3600), + "end": int(time.time() + 60), + }, + ) + assert {expected["submission_trace_id"], expected["execution_trace_id"]} <= { + row["traceID"] for row in job_traces["traces"] + } + metrics = get( + bases["prometheus"], + "/api/v1/query", + {"query": 'last_over_time({__name__=~"inkcre_lab.*"}[1h])'}, + ) + rows = metrics["data"]["result"] + for name, value in { + "inkcre_lab_jobs_total": 3, + "inkcre_lab_duration_seconds_count": 2, + "inkcre_lab_duration_seconds_sum": 0.4, + }.items(): + matching = [row for row in rows if row["metric"]["__name__"] == name] + assert len(matching) == 1 and float(matching[0]["value"][1]) == value, (name, matching) + assert all("inkcre_job_id" not in row["metric"] for row in rows) + buckets = { + row["metric"]["le"]: float(row["value"][1]) + for row in rows + if row["metric"]["__name__"] == "inkcre_lab_duration_seconds_bucket" + } + assert buckets == {"0.1": 1, "0.25": 1, "0.5": 2, "+Inf": 2}, buckets + latency = [] + for _ in range(10): + start = time.monotonic() + get(bases["loki"], "/loki/api/v1/query_range", log_query) + latency.append((time.monotonic() - start) * 1000) + evidence = { + "expected": expected, + "logs": logs, + "traces": traces, + "job_trace_search": job_traces, + "metrics": metrics, + "log_query_roundtrip_ms": latency, + } + (HERE / f"runtime/stack-{action}.json").write_text(json.dumps(evidence, indent=2) + "\n") + print( + json.dumps( + { + "path": action, + "log_to_execution_to_submission": "passed", + "log_absent_vs_zero": "preserved; numeric metadata represented as strings", + "cumulative_counter_and_histogram": "passed; no Job metric label", + "small_log_query_median_ms": round(statistics.median(latency), 2), + }, + indent=2, + ) + ) + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tasks/observability-foundation/experiments/tempo-lab.py b/tasks/observability-foundation/experiments/tempo-lab.py new file mode 100644 index 0000000..190b311 --- /dev/null +++ b/tasks/observability-foundation/experiments/tempo-lab.py @@ -0,0 +1,126 @@ +"""Task-only Tempo replacement probe, reusing the bounded Collector and SSH Docker +transport.""" + +import json +from pathlib import Path +import subprocess +import sys + +from lab import docker, HERE, HOST, IMAGES + +PROJECT = "inkcre-o11y-tempo-g1-b0a97f7c" +SOCKET = "/tmp/inkcre-o11y-tempo-b0a97f7c.sock" +IMAGE = ( + "grafana/tempo@sha256:3076b8dcdfb32fd6bc5ccef85e7b7313e6199b9cb84366257fc17ecb696db5fd" +) + + +def compose(): + collector = { + "receivers": {"otlp": {"protocols": {"http": {"endpoint": "0.0.0.0:4318"}}}}, + "processors": { + "memory_limiter": {"check_interval": "1s", "limit_mib": 384, "spike_limit_mib": 64} + }, + "exporters": { + "otlp_http": { + "endpoint": "http://tempo:4318", + "timeout": "2s", + "retry_on_failure": { + "initial_interval": "1s", + "max_interval": "2s", + "max_elapsed_time": "10s", + }, + "sending_queue": {"sizer": "bytes", "queue_size": 1048576, "num_consumers": 1}, + } + }, + "service": { + "pipelines": { + "traces": { + "receivers": ["otlp"], + "processors": ["memory_limiter"], + "exporters": ["otlp_http"], + } + } + }, + } + return json.dumps( + { + "services": { + "tempo": { + "image": IMAGE, + "cpus": 1.5, + "mem_limit": "3584m", + "ports": ["127.0.0.1:33200:3200"], + "command": ["-target=all", "-config.file=/etc/tempo/lab.yaml"], + "volumes": ["data:/var/tempo"], + "configs": [{"source": "tempo", "target": "/etc/tempo/lab.yaml"}], + }, + "collector": { + "image": IMAGES["collector"], + "cpus": 0.5, + "mem_limit": "512m", + "ports": ["127.0.0.1:34318:4318"], + "command": ["--config=/etc/otelcol/lab.yaml"], + "configs": [{"source": "collector", "target": "/etc/otelcol/lab.yaml"}], + }, + }, + "configs": { + "tempo": {"content": (HERE / "tempo.yaml").read_text()}, + "collector": {"content": json.dumps(collector)}, + }, + "volumes": {"data": {}}, + } + ) + + +def main(action): + if action == "up": + print(docker("compose", "-p", PROJECT, "-f", "-", "up", "-d", data=compose())) + if not Path(SOCKET).exists(): + subprocess.run( # noqa: S603 + [ + "ssh", + "-M", + "-S", + SOCKET, + "-fNT", + "-o", + "BatchMode=yes", + "-o", + "ExitOnForwardFailure=yes", + "-L", + "127.0.0.1:33200:127.0.0.1:33200", + "-L", + "127.0.0.1:34318:127.0.0.1:34318", + HOST, + ], + check=True, + ) + elif action == "restart": + print(docker("restart", PROJECT + "-tempo-1")) + elif action in {"stop", "remove"}: + args = ["stop"] if action == "stop" else ["down", "--volumes"] + print(docker("compose", "-p", PROJECT, "-f", "-", *args, data=compose())) + if Path(SOCKET).exists(): + subprocess.run(["ssh", "-S", SOCKET, "-O", "exit", HOST], check=True) # noqa: S603 + elif action == "logs": + print( + docker("compose", "-p", PROJECT, "-f", "-", "logs", "--tail", "35", data=compose()) + ) + elif action == "stats": + print( + docker( + "stats", + "--no-stream", + "--format", + "{{json .}}", + PROJECT + "-tempo-1", + PROJECT + "-collector-1", + ) + ) + else: + raise SystemExit("Use up, restart, stop, remove, logs, or stats") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tasks/observability-foundation/experiments/tempo-probe.py b/tasks/observability-foundation/experiments/tempo-probe.py new file mode 100644 index 0000000..70f4f6f --- /dev/null +++ b/tasks/observability-foundation/experiments/tempo-probe.py @@ -0,0 +1,116 @@ +"""Reuse the OpenObserve discriminator sample; query before and after a separate restart.""" + +import base64 +import json +from pathlib import Path +import sys +import time +import urllib.request +import urllib.parse + +HERE = Path(__file__).resolve().parent + + +def read_trace(trace_id): + request = urllib.request.Request( + "http://127.0.0.1:33200/api/v2/traces/" + trace_id, + headers={"Accept": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=20) as response: + return json.load(response) + + +def attributes(items): + return {item["key"]: item["value"] for item in items} + + +def compare(expected, response): + spans = [ + span + for resource in response["trace"].get("resourceSpans", []) + for scope in resource.get("scopeSpans", []) + for span in scope.get("spans", []) + ] + assert len(spans) == 1, (expected["name"], response) + actual = spans[0] + assert actual["name"] == expected["name"] + # Tempo's query JSON uses protobuf base64 bytes for IDs; OTLP/HTTP JSON uses hex. + for field in ["traceId", "spanId"]: + assert actual[field] == base64.b64encode(bytes.fromhex(expected[field])).decode() + for field in ["startTimeUnixNano", "endTimeUnixNano"]: + assert actual[field] == expected[field] + assert attributes(actual["attributes"]) == attributes(expected["attributes"]) + assert len(actual.get("links", [])) == len(expected.get("links", [])) + flags_preserved = True + for original, stored in zip( + expected.get("links", []), actual.get("links", []), strict=True + ): + for field in ["traceId", "spanId"]: + assert stored[field] == base64.b64encode(bytes.fromhex(original[field])).decode() + assert stored["traceState"] == original["traceState"] + assert attributes(stored["attributes"]) == attributes(original["attributes"]) + flags_preserved &= stored.get("flags", 0) == original.get("flags", 0) + return flags_preserved + + +def main(action): + input_path = HERE / "runtime/tempo-input.json" + if action == "send": + payload = json.loads( + (HERE / "evidence/openobserve-20261001/ai-projection.json").read_text() + )["input"] + now = time.time_ns() + for span in payload["resourceSpans"][0]["scopeSpans"][0]["spans"]: + span["startTimeUnixNano"] = str(now) + span["endTimeUnixNano"] = str(now + 1000) + span["attributes"].append({"key": "inkcre.job.id", "value": {"intValue": "42"}}) + input_path.write_text(json.dumps(payload, indent=2) + "\n") + request = urllib.request.Request( + "http://127.0.0.1:34318/v1/traces", + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + with urllib.request.urlopen(request, timeout=5) as response: + assert response.status == 200 + print("Synthetic sample accepted by Collector; backend readback pending") + elif action in {"before", "after"}: + payload = json.loads(input_path.read_text()) + spans = payload["resourceSpans"][0]["scopeSpans"][0]["spans"] + output = {span["name"]: read_trace(span["traceId"]) for span in spans} + (HERE / f"runtime/tempo-{action}.json").write_text(json.dumps(output, indent=2) + "\n") + flags = [compare(span, output[span["name"]]) for span in spans] + params = urllib.parse.urlencode( + { + "q": "{ span.inkcre.job.id = 42 }", + "start": int(time.time() - 3600), + "end": int(time.time() + 60), + } + ) + with urllib.request.urlopen( + "http://127.0.0.1:33200/api/search?" + params, timeout=20 + ) as response: + search = json.load(response) + assert {row["traceID"] for row in search["traces"]} == { + span["traceId"] for span in spans + } + (HERE / f"runtime/tempo-search-{action}.json").write_text( + json.dumps(search, indent=2) + "\n" + ) + print( + json.dumps( + { + "phase": action, + "traces_found_by_job": len(search["traces"]), + "attribute_presence_types_and_values": "passed", + "link_ids_state_attributes": "passed", + "link_flags_preserved": all(flags), + }, + indent=2, + ) + ) + else: + raise SystemExit("Use send, before, or after") + + +if __name__ == "__main__": + main(sys.argv[1]) diff --git a/tasks/observability-foundation/experiments/tempo.yaml b/tasks/observability-foundation/experiments/tempo.yaml new file mode 100644 index 0000000..c23abce --- /dev/null +++ b/tasks/observability-foundation/experiments/tempo.yaml @@ -0,0 +1,17 @@ +server: + http_listen_port: 3200 +distributor: + receivers: + otlp: + protocols: + http: + endpoint: 0.0.0.0:4318 +storage: + trace: + backend: local + wal: + path: /var/tempo/wal + local: + path: /var/tempo/blocks +usage_report: + reporting_enabled: false diff --git a/tasks/observability-foundation/inquiry.md b/tasks/observability-foundation/inquiry.md new file mode 100644 index 0000000..76d3dc2 --- /dev/null +++ b/tasks/observability-foundation/inquiry.md @@ -0,0 +1,60 @@ +# 现状证据与待解问题 + +本文件拥有调查结论与未决问题。[design.md](design.md) 消费这些事实,[decisions.md](decisions.md) 保存已经接受的选择;不得把候选方案改写成已观察行为。 + +## 证据边界 + +2026-10-03 复核未变化的源码基线:core-py `1385e6066336eee37626e3b4f4bf68155f9bf1b7`,client-web `367ace98bfb7e9880c002e7971eb397a25c9ef84`,Hub 与 core-py 共享引用均为 `42f7bad1c61e57b5e0ebf55e27815ddc2ae913fa`。Hub 与共享挂载未发现未提交修改;client-web 有与本任务无关的未跟踪 `.agents/skills/vue-ts-code/`,须保留。SVC 状态健康。 + +E1—E12 是定向源码调查;新增实验和官方文档事实见 E13—E28。源码、引用、依赖或部署版本变化后重验受影响结论,合成实验不替代业务验收。首轮 advisor 支持标准传播、独立 Job 关联和可替换后端;此前一次咨询因额度未完成,后续针对实际后端结果的咨询已完成,支持否决 OpenObserve 及限定 Tempo flags 例外。用户明确 serverless/scale-to-0 后,advisor 根据新证据支持 SaaS 优先、OpenObserve Cloud 先验与必要语义门槛,原自建推荐由 D8 替代;随后零费用要求又由 D9 将首验改为 Grafana Cloud Free,并限定 Cloudflare 的补充范围,advisor 支持该收束。选择仍由主 Agent 负责。 + +| ID | 已观察事实 | 主要证据与影响 | +| --- | --- | --- | +| E1 | 一个 deployment 是一个 owner,Peer 可直接操作数据库或委派同步能力 | [共享 authority](../../docs/_shared/20-product-tdd/system-state-and-authority.md)、[Peer 契约](../../docs/_shared/20-product-tdd/semantic-retrieval-and-peer-capabilities.md);不能假定固定前后端或自动增加租户模型 | +| E2 | 日志走 stdout,Python backend 缺省 PG、可选 Logtail/none;当前依赖无 OTel | [日志初始化](../../libs/obsrv/main.py)、[依赖](../../pyproject.toml);保留本地诊断,新增受限的结构化事件出口 | +| E3 | 请求、周期任务和 Job 使用自定义关联 ID | [middleware](../../app/middleware.py)、[scheduler](../../app/scheduler.py)、[JobManager](../../app/business/job.py);不是标准跨 Peer Trace | +| E4 | PG writer 有界排队、批量独立事务,仍共用业务连接池 | [handler](../../libs/obsrv/log_handler_postgresql.py)、[uow](../../app/persistence/observability/uow.py);队列 1024、每批最多 100、排空最多 5 秒,不能称无损 | +| E5 | Web Job 日志页按 `job.` 查 logs 表 | [Job 页面](../../../client-web/apps/client-web/src/views/jobs/job/job.vue)、[Log 查询](../../../client-web/packages/core/src/obsrv/log.ts);不能原位替换旧 trace_id 后直接退役 | +| E6 | 浏览器通过 PostgREST 提交、领取和关闭 Job,也能本地执行 handler | [TS JobManager](../../../client-web/packages/core/src/job/manager.ts)、[Job schema](../../../client-web/packages/core/src/job/job.ts);跨语言生产者和执行者都属于契约消费者 | +| E7 | Python 手动任务与 Cron 复用 create_in_uow;当前 Job 无提交上下文字段 | [Python JobManager](../../app/business/job.py)、[CronManager](../../app/business/cron.py)、[Job schema](../../app/schemas/job.py);只在 REST middleware 加上下文会漏路径 | +| E8 | Peer ID 已有 UUID;通用 deployment configs 已存在,但检查的配置/身份表面未提供统一遥测部署标识 | [Peer schema](../../app/schemas/peer/main.py)、[settings](../../app/settings.py)、[configs](../../app/schemas/deployment_config.py);选择标识来源时不推导数据库 URL、凭据或后端项目 ID | +| E9 | Agent debug 默认关闭,已有 thread/turn/model/tool 事件 | [debug](../../app/business/agent/debug.py)、[thread](../../app/business/agent/thread.py)、[说明](../../docs/40-deployment/agent-debug.md);适合复用采集位置,不是执行恢复记录 | +| E10 | 两类 chat adapter 都未保留 usage;阿里流式路径跳过无 choices 的块 | [OpenAI-compatible](../../app/business/ai/dialects/openai_compatible.py)、[Alibaba](../../app/business/ai/dialects/alibaba_model_studio.py);usage/流结束的准确性需在 provider 边界验证 | +| E11 | Thread 存于内存,Agent Query 结果已有 answer/references | [Thread persistence](../../app/business/agent/persistence.py)、[Agent Query](../../docs/30-unit-tdd/agent-query-sink.md);不能把 Trace 当结果或恢复权威 | +| E12 | 现有 Compose 声明 PostgreSQL、core、PostgREST;开发数据库由 SVC 管理 | [Compose](../../docker-compose.yml)、根 `AGENTS.local.md`;观测实验需独立资源,不借清理实验删除现有数据库 | +| E13 | 标准 Python/JS propagator 互操作通过;Python malformed tracestate WARNING 含原始 canary | [实验报告](experiments/README.md);浏览器尚未运行,基础远端来源限制必须覆盖 SDK 诊断 | +| E14 | 旧 Python/TS Job 模型容忍新增列,旧 REST form 拒绝额外字段 | [旧消费者实验](experiments/README.md);该轮仅模型边界,后续实际 SQL 更新见 E18 | +| E15 | core-py ready 要求 migration heads 精确相等 | [readiness](../../app/database_contract/readiness.py)、[bootstrap](../../run.py);不得由 nullable 列推导旧 runtime 在新 schema 获准运行 | +| E16 | OpenObserve 合成三信号可查询,但丢失 AI absent/zero 区别;中断时 Collector 有界丢失 | [原始实验](experiments/README.md);后端候选重开,不宣称业务故障隔离已通过 | +| E17 | Tempo 重启前后保留四条样本的当前 AI/因果语义,历史 Link flags 丢失 | [Tempo 证据](experiments/README.md);当前语义通过,完整 OTLP 保真未通过,全栈后续见 E20 | +| E18 | disposable PG 上实际旧 Python repository 和 TS JobManager/DBAPIClient 经 PostgREST 创建默认 NULL,claim/close 保留 carrier;模拟新 head 时实际旧 readiness 拒绝 | [数据库证据](experiments/convergence-20261003.md);支持两列与协调升级,不等于正式 migration/完整启动已验 | +| E19 | SDK traceparent 注入 55 bytes,合法 tracestate 可达 1109 bytes;数据库拒绝 513-byte 字段 | [容量证据](experiments/convergence-20261003.md);512 是应用存储政策,capture 超限 optional state 整体省略 | +| E20 | 五组件独立 API、Grafana proxy 和 UI 跑通 Job 日志→execute/model→submit Link;Loki 保持 absent/zero,Prom counter/histogram 正确 | [闭环证据](experiments/convergence-20261003.md);Editor 完整导航,Viewer 只有固定面板 | +| E21 | 2 CPU/4 GiB 限制下查询后的 Docker 快照约 608 MiB;Grafana 首次启动约 8 分 18 秒 | [资源证据](experiments/convergence-20261003.md);非稳态/峰值容量,生产健康宽限与预算须 G3 验 | +| E22 | 现有自托管 Compose 在 core-py;未发现独立 infra 仓库。client-webext 是独立旧 direct-fetch repo,iOS/rokid 有各自原生入口,ext-reg 为 Worker | [覆盖与归属](task-map.md);原自建配方归属依据;不代表应当自建,也不能宣称共用 Web 接入自动覆盖所有客户端 | +| E23 | OO1.0.4 原始 -1 保留,但 total 被补为0/部分和,省略 cost 仍补零;source=unavailable 保留,直接负值求和错误 | [本轮实验](experiments/sentinel-saas-20261003.md);允许最少来源元数据与受控查询,不再以补值全盘否决平台 | +| E24 | OO Cloud 按量收费;Grafana Cloud 有免费三信号和标准直发;PostHog 原生 OTLP 三信号齐备但 metrics Alpha、tracing Beta、AI/Trace 查询和 UI 分离 | [官方资料判别](experiments/sentinel-saas-20261003.md);无需自建常驻栈,目标账户仍待准入 | +| E25 | Sir 明确 serverless、scale-to-0 和 SaaS 优先,偏好 OpenObserve | 当前会话;否定原单机/常驻 Collector 的默认预算假设,D8 与 V8 已修订 | +| E26 | Sir 暂时不能使用收费服务;Cloudflare 原生观测/AI Gateway/Analytics Engine 有免费能力,但通用多 Peer 后端仍有范围缺口 | [Cloudflare 与免费方案](experiments/sentinel-saas-20261003.md#cloudflare-与当前零费用方案);D9 改为 Grafana Cloud Free 首验,按日期区分 Cloudflare 新旧限额 | +| E27 | Python ObsrvSetting 默认 postgresql,.env.example 同样为 PG;Compose 未配置 fallback 为 none | [setting](../../libs/obsrv/setting.py)、[初始化](../../libs/obsrv/main.py)、[Compose](../../docker-compose.yml);D10 保留默认 PG,实施时对齐 Compose fallback 并保留显式覆盖 | +| E28 | Sir 同意 Grafana Cloud Free,强调 vendor-agnostic、新遥测按需开启且默认保留 PG 日志 | 当前会话及 [D10](decisions.md);撤销 PG writer 退役与端点存在即启用,新增 V0、修订 V3/V7 | + +## 问题处置 + +| ID | 状态与处置 | 对后续的影响 | +| --- | --- | --- | +| Q1 | 尚缺 Sir 的原文/长期复核/评测承诺;新增 OTLP 按运行诊断+结果来源关联、原文默认关闭定稿;原 PG 内容行为保留 | 不阻挡元数据实现;开启长期快照前确定保存/删除与结果 owner schema,不能把 V10 擅自排除 | +| Q2 | SaaS/serverless/scale-to-0 与当前零新增观测费已明确;实际事件/查询量、区域和保留需求仍未给出 | 先验实际 Free 的必要能力和额度;不能用低单价/试用绕过零费用约束 | +| Q3 | 当前按一个 deployment 一个 owner;未出现跨 owner 集中运营需求 | 如出现则重开数据/查询边界,不从部署属性推导隔离 | +| Q4 | 已收敛:[共享形状](design/shared-contract.md)确定两列、容量、config key/schema 和初始化责任 | 进入 Hub/数据库协议实现;保持 exact-head 与协调升级 | +| Q5 | D10:Sir 已同意 Grafana Cloud Free 按需启用;Cloudflare 用于实际覆盖部分,目标 SaaS 准入未运行 | 仅验必要查询/字段/入口,失败才转备选;账户/预算边界仍归 Q2 | +| Q6 | 已修订:各 Peer 本地 opt-in 后直发 SaaS,默认 PG 不变;部署 owner 管理项目与数据生命周期,转发/Collector 按缺口引入 | 无需新平台仓库或默认常驻主机;core-py 分发相关接入文档 | +| Q7 | 源码覆盖清单已完成,活跃发布/运行实例仍需部署盘点 | 所有已发现 Unit 留在 G3 清单,未知运行状态不算不适用 | + +此前 Q1/Q2 未收到具体参数而采用单机假设;Sir 本次已明确修正 Q2 的首要标准,现按 SaaS 与 scale-to-0 推进。本次零新增观测费已由 Sir 明确,长期内容承诺仍不能从未答复推导,生产启用在对应交付门槛收束,不重复索取总体架构许可。 + +## 新增共享 truth 的拟归位 + +身份、传播、Job 上下文和观测/业务 authority 的区别需要 Python 与 TS 共同理解,拟由 Hub `20-product-tdd/` 的独立观测契约承载,现有 cross-unit 导航与 Job 契约仅引用或补充必要连接。SDK 初始化、logger wiring、UI 组件、Collector 配置和资源值留在各 Unit/部署 owner。 + +当前 Hub 可用且与 Spoke 引用一致;没有理由从 `docs/_shared/` 修改。需遵循 [共享文档工作流](../../docs/_shared/00-meta/skills/edit-svc-shared-docs/SKILL.md) 的 Hub 源先发布、Spoke 引用后推进顺序。已准备[未应用 Hub patch](design/hub-contract.patch)及[评审说明](design/hub-review.md)。当前源仓与挂载未修改;正式发布仍需明确提交/推送指令。 diff --git a/tasks/observability-foundation/packet.md b/tasks/observability-foundation/packet.md new file mode 100644 index 0000000..c40a916 --- /dev/null +++ b/tasks/observability-foundation/packet.md @@ -0,0 +1,36 @@ +# InKCre 可观测性基建 + +建立适合多 Peer、serverless 与 scale-to-0 的可选观测能力。Sir 已选定 Grafana Cloud Free,要求 vendor-agnostic、默认维持 PostgreSQL 日志,并已授权实施、创建分支、提交、推送与 draft PR。没有合并、生产发布或付费服务授权。 + +## 当前交付 + +首批实现已进入集成收尾:Python 与 client-web 使用标准 OTel/OTLP,本地开关缺省关闭,PG writer 与 Job 查询保留。Hub 契约已发布到分支;两 Spoke 的共享引用各自独立提交。两个 Spoke 的源码均已提交、推送并发布 draft PR;任务包与实验归档作为独立提交收尾。 + +| 对象 | 当前状态 | +| --- | --- | +| Hub | `codex/observability-contract`,`a603b036fd41dae5a471dea24cd7609bb4927bd5`,[draft PR #33](https://github.com/InKCre/docs/pull/33) | +| core-py | [draft PR #124](https://github.com/InKCre/core-py/pull/124),源码 `0eba26a`,共享引用提交 `f0397e7`,基于 `fa23a1c`;默认关闭、三信号、HTTP/Peer/Job/Cron/AI/Agent/来源事件、正式迁移及受控 protobuf relay | +| client-web | [draft PR #125](https://github.com/InKCre/client-web/pull/125),源码 `b81554d`,共享引用提交 `b1072a9`,基于 `d4c6437`;连接本地开关、三信号、Peer/Job carrier/Link、可选诊断链接与认证 relay | +| Grafana Cloud | Sir 已创建 stack 并配置写入与 Viewer 凭据。三类合成数据已由 Viewer API 独立读回;Trace/Span/Link、AI unknown/zero、日志和指标均与源输入一致 | + +持久行为见 [core-py 部署文档](../../docs/40-deployment/observability.md)和已发布[共享契约](../../docs/_shared/20-product-tdd/observability-contract.md)。任务内旧设计与实验保留其历史语境,不再作为运行事实的第二权威。 + +## 已验证与收尾 + +真实隔离 PostgreSQL 已迁移至 `3d9593b0c855`,Python Job 成功/失败/取消与 PG 日志、mixed 开关、三信号及 metrics-only 已验。实际 FastAPI lifespan/readiness/JWT relay 已验。实际 OpenAI SDK 对合成 provider 的流式末块、unknown/zero、并行工具与内容 canary 已验,不等于真实付费模型验收。 + +真实 Chromium+PostgREST 验证了直写、执行 Link、异步并发、重定向、JWT 与三信号。真实浏览器→core relay→接收端逐值核对 Trace/Span/Link IDs 已通过;此前 JSON 返回 200 的证据作废,通用 ProtoJSON 会误解 OTLP 十六进制 ID,因此采用官方 protobuf exporter,并明确拒绝 JSON。 + +- [本轮验收](verification.md)管理具体主张、证据和残余;[客户端报告](experiments/client-implementation.md)给出客户端范围。 +- SDK 故障隔离、部分初始化资源回收、原生队列/导出计数已通过独立复验;迁移升级/降级/再升级保留旧 Job/PG 日志并验证容量约束。最终 `pdm run check` 已通过(14 passed、63 skipped,0 类型诊断)。 +- core 最终 `pdm run check` 通过(14 passed、63 skipped)。client lint、全量类型检查、build 与 tracked 文件格式通过;完整 `pnpm check` 被用户原有未跟踪 skill 格式阻塞,保持该文件原状。 + +## 剩余交付门禁 + +本批 draft 不等于完整 G2/G3 或生产准入。正式客户端类型需在 core release artifact 准入后,使用既有 sync 流程重新生成并比较;不修改准入规则来容纳候选数据库。Hub 合并后依赖 Spoke 必须更新至合并后的共享 SHA。 + +云端合成数据读回及 API 样本导出已通过,免费额度、保留边界与长期历史导出仍待验证;真实平台终止宽限与请求后冻结、真实模型 provider、活跃部署覆盖及 webext 独立生命周期仍是部署或后续覆盖门禁。AI 原文/长期快照没有被基础 tracing 自动授权。正常 SDK shutdown 的等待窗口不是端到端硬时限或零丢失承诺。 + +## 资源与权限 + +本轮 task-owned `inkcre-o11y-db-g1-b0a97f7c` 的两个容器和隧道已停止;`o11y_impl`、`o11y_migration_roundtrip`、历史 `o11y_lab` 与其余实验卷保留。其它三历史实验项目也保持停止。未操作 SVC 开发数据库;未升级套餐、合并或生产部署。真实凭据仅留本地忽略配置,不进入任务证据、提交或浏览器。父任务保持活跃。 diff --git a/tasks/observability-foundation/task-map.md b/tasks/observability-foundation/task-map.md new file mode 100644 index 0000000..7e239c9 --- /dev/null +++ b/tasks/observability-foundation/task-map.md @@ -0,0 +1,74 @@ +# 工作边界与交付顺序 + +> 当前执行进度:G1 已收敛并授权实施,Hub draft PR #33 已发布,core/client 首批实现与本地 G2 已运行,正在云端合成读回与提交收尾。本文保留全父任务的工作地图;旧“未实施”状态由 [packet](packet.md) 与 [verification](verification.md) 的当前记录替代。 + + +[packet.md](packet.md)是唯一 Human 入口。本文件拥有跨仓依赖和实施顺序;[整体设计](design.md)拥有架构,[共享契约形状](design/shared-contract.md)拥有字段/配置/兼容,[验收表](verification.md)拥有证据。各仓不复制第二个父任务包。 + +## 阶段和责任 + +| 阶段 | 状态与退出条件 | +| --- | --- | +| G1:契约与可行性 | 业务契约已收敛;D10 已确认 Grafana Cloud Free、显式 opt-in 与默认 PG;目标 SaaS 准入待验。既有传播、旧消费者/准入、carrier 和本地三信号证据保留;目标 SaaS 尚未验证 | +| G2:真实多 Peer 集成 | 待实现。契约按 Hub→Spoke 发布;Python/浏览器提交与执行、同步委派、Cron、模型/工具形成可查询因果关系;旧查询、出口内容边界和业务故障隔离通过 V1—V6 | +| G3:覆盖与运行交付 | 待实现。活跃 Unit 逐项完成,实际用量/配额/保留/导出/恢复边界、运维入口和回退通过 V7—V9;V10 依确认后的长期证据承诺验收 | + +四条工作线持续存在:C 拥有共享契约与兼容;I 拥有托管接入和运行;P 拥有各 Peer 采集/PG 日志兼容;A 拥有 AI 调用与业务结果证据。主 Agent 负责全局集成和 Human 沟通,独立工作只有在明确低耦合收益时委派。G1 两个单元的返回见 [C-G1](cells/contracts-foundation.md)、[I-G1](cells/infra-foundation.md),随父任务保留,不因子单元完成而清理。 + +## 可执行的实施顺序 + +### 1. 发布共享契约和机械协议 + +变更对象是 Hub 的四文件 patch:无统一观测语义 → 公共身份/传播/Job carrier/内容边界。影响所有参与 Peer,不改变业务 authority。先审查并在 Hub 源交付,经明确指令提交/推送后,各 Spoke 分别更新共享引用;Hub 修改、引用 bump 和本地实现不得混为一个提交。当前 patch 仍只在任务内,尚未应用。 + +core-py 数据库协议 owner 随后交付两列、容量约束、config schema 注册、数据库初始化步骤与生成协议投影;client-web 同步消费生成类型。实际 migration 在 disposable DB 验证旧记录/旧消费者、约束和 exact-head。新增遥测默认关闭且不读取其共享配置;未配置或关闭不影响 PG 日志和应用启动。新 Job 字段依赖对应 schema admission,不靠写失败重试。正式发布采用共享契约已描述的协调升级,不能把 nullable 推导为无停机滚动兼容。 + +### 2. 验证并交付 SaaS 接入 + +先用一个 Grafana Cloud Free 目标账户确认非试用/无付费依赖,再验证三信号、Job 因果查询、AI unknown/zero/partial 的写入/查询/导出、实际诊断入口和预算控制;不上传真实内容。只有必要条件失败才重看其它免费候选,不同时建设多家生产出口;OpenObserve Cloud 暂缓。当前没有云端账户运行证据;不采购、不自动升级、不向第三方发送真实内容。 + +core-py 的 `docs/40-deployment/observability.md` 拟拥有启用、身份、权限、费用/限额、保留、导出与切换步骤;各 Peer 交付默认 false 的本地开关、标准 SDK 和按信号出口配置;endpoint/凭据不得自动启用。没有默认的 `deploy/observability/` 五组件交付项。部署 owner 管理供应商项目,各运行平台保管本地私密凭据。 + +验证进程直接 OTLP、短生命周期 flush 上限和计费影响;指标用 push,不用 scrape 唤醒归零实例。浏览器只使用经验证的客户端写入能力,否则通过受控按请求转发;查明 CORS、权限、限流和失效行为。仅在供应商或现有平台不能承担必要机制时引入 Collector,并记录具体缺口与运行成本。 + +接入失败不阻止应用 ready,不重试业务事务。费用以实际写入、查询、保留和应用出口计量;用量上限对应告警、限流还是丢弃必须实测,不承诺所有供应商都有硬费用上限。完成必要准入即进入真实业务切片,不把无限产品比较当交付。 + +### 3. 先验证默认行为,再完成显式开启的跨 Peer 切片 + +Python 的 `libs/obsrv` 保留现有 logging_backend 与 PG 生命周期,新增独立本地 telemetry_enabled;Compose 未配置 fallback 从 none 对齐 PostgreSQL,但保留显式覆盖。先验默认关闭及“已有 endpoint/凭据仍关闭”时没有新增 OTLP/provider/队列而 PG/Job UI 正常,再验开启后的标准 SDK 初始化与有界关闭、Resource、专用结构化日志来源。middleware/Peer 调用负责受控 HTTP context;Job/Cron 的共用创建入口写 carrier、领取后的执行 scope 建 Link。client-web 共用 core 包负责 JobManager、Peer HTTP 和 DBAPIClient,Web 页面负责诊断入口。 + +明确为参与测试的 Peer 开启新遥测,以“浏览器 Peer A 提交 Job → Python Peer B 执行 Agent → 能力调用 Peer C → 模型和并行工具 → 结果与引用”为首条旅程;反向增加 Python 提交、浏览器实际执行。覆盖开→关、关→开、关→关、开启后再关闭,以及发起方重启、NULL/损坏 carrier、Cron、并发隔离、取消、失败、结果未知及关闭。标准事件至少区分提交、成功 claim/开始、终态、能力未执行/已执行/结果未知;只在业务已确认这些事实的边界发出,不能从日志顺序推断状态。 + +AI adapters 在 provider 回包位置采集使用量,覆盖流式 usage-only 末块、embedding、无 usage 和真零;Agent/Tool/检索只记录固定事件和受控 ID。先用受控回包比较,再在授权 preview 用实际 provider 验证外部行为。价格估算由版本化价格/币种配置负责,未知价格不记零;不用 tracing 代替账单。 + +同一业务负载断开 SDK→SaaS,以及实际存在的转发/Collector→SaaS,比较原有 API/数据库终态、内存/队列、关闭时限和丢弃计数;遥测失败不能让 Job 创建重试或业务 ready 失败。内容 canary 从请求、异常、SDK warning、SQL 参数和工具结果进入,检查 Peer 第一次发送的 OTLP,不能只检查后端清洗结果。 + +### 4. 保留 PostgreSQL 日志,添加可选诊断入口 + +Job 页当前及历史 `job.` 查询、分页与排序继续使用 PG 路径,不迁移或替换其 trace_id/span_id。新后端链接只作附加入口,未开启、无可用 Trace 或已经过保留期时 PG 日志仍可查询。关闭新增遥测不需要回切 PG,不删除历史,不回填此前未采集的 Trace。 + +新增 OTLP logger 不自动桥接 PG 历史、任意应用日志或 agent_debug 内容;现有 PG/Logtail 的内容行为与配置保持。撤销旧 writer 退役计划;未来是否停写或删除日志属于另一个明确授权的变更。exact-head 仍使旧二进制不能直接配新 schema,关闭遥测不等于数据库 downgrade。 + +后端可替换验收只改标准出口及认证配置,把同一采集样本交给另一兼容接收端,核对必要 ID/Links/unknown 与聚合;业务采集和持久字段不改。Grafana 查询、datasource、面板和深链接在部署/消费侧维护,不引入通用插件框架,也不声称消费侧零迁移成本。 + +### 5. 扩面与运行验收 + +按下表确认活跃部署,补齐采集和查询;没有活跃运行证据的 Unit 保持“待部署盘点”,不能自动宣布不适用。G3 记录每天摄取/查询量、指标序列数、保留、实际费用、峰值和查询 P95;验证供应商配额、删除、导出与配置重建,明确托管服务能提供的恢复范围。独立管线探活不能唤醒归零业务实例;通知目的地由部署 owner 配置,任务不自动给外部联系人发消息。若以后选择自建,再加入磁盘/冷备恢复与启动资源验收。 + +| Unit / 入口 | 接入责任和范围 | 当前证据与验收 | +| --- | --- | --- | +| core-py / 进程内 extensions | 共用 Python SDK 出口,HTTP、Peer、Job、Cron;Source/Sink/Organization/Resolver/Storage 在业务边界补语义 | 源码已定位;G2 首批实际运行 | +| client-web 共用 core + Web app | Peer/Job/DBAPIClient、浏览器 context、Job UI | 实际 TS/PostgREST 旧消费者仅在 Node 验证;浏览器仍须验收 | +| 独立 client-webext | [root.ts](../../../client-webext/logic/root.ts)、[block.ts](../../../client-webext/logic/block.ts) 的旧直连 HTTP 路径单独接入 | 没有证据它消费当前共用 core;不能称 Web 接入自动覆盖它,也不借本任务迁移其业务协议 | +| client-ios | [APIClient](../../../client-ios/InKCre/APIClient.swift)、[NetworkService](../../../client-ios/InKCre/NetworkService.swift) 的 URLSession 边界 | 活跃发布和现行协议待盘点;采用环境适配的标准传播/导出,不先强加 Job 执行身份 | +| rokid-studio-client | 相机/OkHttp 上传边界,默认只采元数据 | 旧业务 endpoint/OSS 路径的原生应用;活跃部署待盘点,不采图像或签名地址 | +| PostgreSQL / PostgREST / 主机 | 部署 owner 使用已有免费指标/日志出口;有内容的数据库日志单独启用 | 客户端 span 不算 DB server span;不改变 SVC 数据库生命周期 | +| 官方 ext-reg | Worker 与 D1/R2 的现有边界,由该服务运营 owner 采集 | 单独部署/数据归属;选 Worker 兼容标准出口或平台原生采集,不盲装 Node SDK;现成 OTLP 导出仅 logs/traces,metrics 缺口单独验收,不把各用户部署合并为一个 owner | + +本顺序没有删除原始“整个 InKCre”范围。AI 先按诊断与来源关联设计;若 Sir 要求所有历史答案/图谱变更都可复核当时内容,V10 必须加入业务结果 owner 的持久快照实现,不以关闭内容开关宣布该要求完成。 + +## 关闭条件与授权 + +当前已完成业务契约与 SaaS 方向修订,目标云端准入尚待验证;源码实现、提交/推送和正式部署未执行。实现授权后,应持续完成 G2/G3 所需代码、测试与文档,不能再以方案代替交付。各仓检查按其本地治理执行;协议/数据库与外部行为分别使用 disposable runtime 和授权 preview。 + +父任务最终关闭需真实验收、持久文档归位、跨仓引用与回退关系明确,实验资源与历史数据的保留/删除有依据。既有 G1 子项完成与本次方向修订均不关闭父任务,任务包和合成数据卷继续保留。 diff --git a/tasks/observability-foundation/verification.md b/tasks/observability-foundation/verification.md new file mode 100644 index 0000000..dbf4aa2 --- /dev/null +++ b/tasks/observability-foundation/verification.md @@ -0,0 +1,53 @@ +# 跨工作线验收 + +本文件汇总整个任务的交付主张、判别性证据和残余;预期行为来自已接受方向与已发布到 Hub draft PR #33 的共享契约。它不替代各仓检查,也不凭测试通过授予发布或范围缩减权限。 + +## 主张与证据要求 + +| ID | 需要证明的结果 | 判别方法与观察来源 | 当前状态 | +| --- | --- | --- | --- | +| V0 | 默认行为与显式 opt-in 正确 | 默认配置、已有 endpoint/凭据但开关关闭、显式开启、故障、改回关闭后重启/连接初始化五种状态,核对 PG 写入/Job UI;关闭态无新增 provider/后台队列/OTLP 请求或传播捕获。Compose 无值默认 PG,显式 none/logtail 保持 | 已验:关闭时即使配置出口也无新增线程/发送,默认 PG;实际应用两种模式 ready/关闭通过;Compose 默认已对齐 | +| V1 | 同步 Peer 调用因果关系正确,直接数据库访问的盲区被准确呈现 | 用真实浏览器/HTTP 传输发起受控操作,独立后端读回 trace/span/Peer/能力关联;PostgREST 客户端 span 不算服务端 SQL 证据 | 真实 Chromium/Peer 与真实 core relay protobuf 已验;Trace/Span/Link ID逐值一致。Cloud 合成提交/执行/AI 三个 span 与 Link 已独立读回 | +| V2 | 提交与另一 Peer 的 Job 执行可关联,重启不依赖内存上下文 | 提交后重启发起 Peer;另一执行 Peer 领取;核对持久提交上下文、执行 Span Link 和 Job 终态。覆盖 Python/TS 生产者与执行者、Cron、旧 Job 及开→关/关→开/关→关;关闭端创建 NULL,领取/关闭保留已有 carrier | 正式 migration、实际 Python/TS Job和mixed开关已验,关闭端保留carrier。独立迁移升级/降级/再升级及容量约束已验;生产协调升级未执行 | +| V3 | 新增观测开关不改变现有 PG 日志能力 | 开关关闭/开启/后端中断/再次关闭时,从真实 Job 页核对当前与历史 PG 日志、job.、排序分页;新诊断链接为可选,旧 writer 不自动停写,原日志 trace_id 不替换为 OTel ID | Python真实Job/PG日志在开启与关闭路径保持;客户端Job页入口保持并有UI烟测。生产历史数据分页仍待部署验收 | +| V4 | AI 耗时、结束原因、usage 与步骤关系可信 | 对受控非流式、流式末块、usage 缺失和并行工具案例比较 provider 原始回包与导出数据;再用授权 preview 验证一个真实 provider。未知不记零,估算成本不冒充账单 | 实际 OpenAI SDK+合成provider经生产adapter/Agent已验:31span/13HTTP请求,usage-only末块、重复累计、unknown/zero、并发错误取消。真实外部provider尚未调用 | +| V5 | 新增 OTLP 基础模式没有意外内容副本,采集权限不等于读取/管理权限 | 合成敏感 canary 经过请求、异常、工具结果与 SQL 参数路径后,检查实际导出数据;验证采集入口及读者边界。PG 按原配置保留,不能以此宣称整个部署无原文。新增内容模式另验保留、截断、删除与 blob 访问 | 实际OTLP出口canary与受控字段已验;relayJWT、私密头、256KiB/并发4/3秒已验;SDK内部指标View/exemplar去敏已验。旧PG原文行为保持 | +| V6 | 遥测不可用不改变业务成功、失败、取消和资源关闭语义 | 同一受控负载对比正常与断开 SDK→SaaS 和实际存在的中间转发;检查数据库业务终态、响应、队列/内存上限、丢弃计数和恢复;不可用时不能阻止应用 ready | 正常/慢接收/拒连、真实Job成功失败取消、metrics-only已验;8项SDK故障注入通过;1024条积压产生原生 queue_full 计数;已记录SDK并发误差,不能用作精确损失账本,最后metrics回收export失败 | +| V7 | 可更换后端并保留必要的诊断能力 | 仅改标准 endpoint/认证配置,将同一采集样本改投第二个兼容后端,业务采集/Job schema 不改;按 Job/Trace/AI 字段查询,核对 links;另做历史数据导出读回与配置迁移清单。只收到 OTLP 不能算通过 | 历史第二后端实验保留,新增实际浏览器/core三信号标准PBF已验;目标Cloud必要 Trace/Link/AI 字段、日志/metric查询及样本导出已通过;长期历史导出仍未验 | +| V8 | 托管接入符合 scale-to-0 与当前零新增观测费,可维护和迁移 | 验 SDK 直发与短生命周期 flush/冻结、无 scrape 唤醒、应用附加计费时间;确认实际 Free、无收费依赖,量化摄取/查询/保留额度与限额失效,验删除/导出/配置重建和供应商恢复范围。自建时另验冷备与资源 | 实际stack已创建;Viewer查询API三类均200,写入认证已修正,实际三信号存储读回通过。默认无常驻Collector/scrape;真实免费套餐额度及平台freeze/终止宽限未验 | +| V9 | 整个实际运行集合有明确覆盖结论 | 按 Unit/部署列出业务入口、传播、采集和查询证据;Source/Organization/Storage、各客户端及官方运营边界逐项验收或由 Sir 明确范围 | 首批点位已实现,覆盖表见部署文档;webext独立生命周期、非Agent直接来源事件与活跃部署逐项仍未通过 | +| V10 | AI 长期证据符合确认后的产品承诺 | 若纳入:删除/改变当前源数据后仍能按保存契约复核当时输入与引用;核对业务结果 owner、版本与删除语义。Trace 全采样不算持久证据保证 | 维持诊断/来源关联与原文默认关闭;没有承诺或新增长期快照,具体产品需求另行确认 | + +## 当前实现证据(2026-10-03) + +`experiments/evidence/foundation-probe.json`、`runtime-probe.json`、`runtime-probe-metrics-only.json` 与 `application-probe.json` 覆盖真实SDK传输、Job/PG与应用生命周期。`experiments/client-browser-result.json` 与 `client-real-relay-result.json` 覆盖真实浏览器/数据库与最终protobuf relay;旧JSON 200不再构成ID保真证据。AI细节见 `ai-instrumentation-notes.md`。机械检查和Cloud结果在收尾时更新,不将请求200混同于存储读回。 + +## 执行与可信范围 + +首个受控业务旅程是“Peer A 提交 Job → Peer B 执行 Agent → 能力调用 Peer C → 模型/工具 → 结果与引用”,并增加浏览器本地执行这一对等方向。不要通过强制所有操作经过 core-py 构造一条看似完整的链路。 + +业务结果以原有 API/数据库状态及界面为观察来源;导出是否成功以接收端查询读回为依据;Token 以已知 provider 回包为对照。构造回包的脚本和消费代码不能互相抄字段后自证完整,至少覆盖流式 usage-only 末块及 unknown 的真实边界。 + +每次有效运行保留精确源码/镜像/规范版本、拓扑、合成输入标识、查询或 UI 读回、负载参数、结果和残余;原始大日志放在该次实验的任务内目录,入口只保留判断所需摘要。不保存真实凭据和未经授权的生产内容。 + +先运行最窄、能否定主张的检查。共享数据库迁移、并发领取、跨进程传播和故障隔离可需要真实 disposable PostgreSQL 与传输;不增加映射/私有 helper 的镜像测试。仅整理任务包时检查链接、状态归属及 whitespace,不运行全仓 `pdm run check`。 + +## G1 历史收敛与 SaaS 修订验证 + +2026-10-03 较早的数据库/三信号补充实验在原自建假设下曾满足 G1 退出条件:公共形状和 owner 已明确,真实旧消费者/准入已判别,三信号可在独立 API、Grafana proxy 与实际 UI 查询。独立 advisor 复核未发现阻挡进入实现的设计缺口,建议结束选型实验。V0—V10 的任务级实现验收没有因此整体通过。Sir 随后明确 serverless/scale-to-0,D8 重开部署后端准入;本轮 -1/来源标记实验和官方 SaaS 文档只支持修订方向,不替代目标 Cloud 验收。 + +脚本语法、归档 JSON/JSONL、本地 Markdown 链接/围栏、patch 和工作树隔离由最终检查记录于 `experiments/evidence/convergence-20261003/packet-check.json`。Hub patch 对源仓只读 `apply --check`,并在临时副本应用核对四文件范围、链接和生成 SVC 区域。没有为任务包运行全仓 `pdm run check`。 + +前轮独立归档核对确认 AI absent/zero/explicit、4096 spans 丢弃总数、Tempo 重启前后属性/Links;flags 丢失保留为失败项。本轮数据库/独立 API/proxy 由脚本断言,UI 由浏览器可见状态单独核对,资源结束状态另查 Docker 与 SSH sockets,不以脚本退出零替代语义判断。 + +## 重验与结束条件 + +更换 SDK/GenAI semconv、后端版本、字段映射、部署网络、采样/内容策略、Job carrier 或结果保存规则后,重验受影响的 V 项。源码静态结论不能升级为部署证据,开发机结果不能自动升级为生产容量结论。 + +当前证据覆盖真实候选应用+隔离数据库/浏览器,以及目标 Grafana Cloud 的合成样本读回,详见 [Cloud 验收](experiments/cloud-acceptance.md)。这不代表生产业务、真实外部模型或全活跃部署覆盖。G1 已收敛,G2 首批集成已运行;G3 仍需补齐免费额度/保留/平台生命周期和覆盖,V10 依具体产品承诺确定。任何不适用项必须说明条件与决定 owner,不能用未执行或失败代替不适用。 + +任务关闭还要求持久文档归位、各仓交付/回滚关系明确、历史数据与实验资源的去留有依据。仍有依赖用户事实或授权的动作时,报告具体残余,完成独立可推进工作,不把父任务标为完成。 + +本次 SaaS 修订的语法、链接、证据、Hub patch 与资源检查记录在 `experiments/evidence/sentinel-saas-20261003/packet-check.json`;原报告保持历史时间边界,不覆盖前轮证据。 + +历史 D10 阶段只做显式 opt-in、PG 保留和 vendor-agnostic 候选差异的任务包/Hub patch 静态检查;当时未运行新应用实验。当前 V0/V3 结论以本轮实现证据为准。检查见 `experiments/evidence/sentinel-saas-20261003/opt-in-packet-check.json`。 From 4e4b10e8ed1be49d09456bbdd0d7e4267a3c0f17 Mon Sep 17 00:00:00 2001 From: Lan_zhijiang Date: Sat, 3 Oct 2026 20:52:09 +0800 Subject: [PATCH 4/5] =?UTF-8?q?docs:=20=E5=AF=B9=E9=BD=90=E5=B7=B2?= =?UTF-8?q?=E5=90=88=E5=B9=B6=E7=9A=84=E5=8F=AF=E8=A7=82=E6=B5=8B=E6=80=A7?= =?UTF-8?q?=E5=85=B1=E4=BA=AB=E5=A5=91=E7=BA=A6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/_shared | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/_shared b/docs/_shared index a603b03..7916312 160000 --- a/docs/_shared +++ b/docs/_shared @@ -1 +1 @@ -Subproject commit a603b036fd41dae5a471dea24cd7609bb4927bd5 +Subproject commit 7916312c12cfaf2ee396d7dc1afffee67d4f0721 From 1a556d60c717458ba3595d7dc50683fc941219cb Mon Sep 17 00:00:00 2001 From: Lan_zhijiang Date: Sat, 3 Oct 2026 21:23:54 +0800 Subject: [PATCH 5/5] =?UTF-8?q?docs:=20=E8=AE=B0=E5=BD=95=E7=9C=9F?= =?UTF-8?q?=E5=AE=9E=E9=A2=84=E8=A7=88=E5=88=B0=20Cloud=20=E7=9A=84?= =?UTF-8?q?=E9=AA=8C=E6=94=B6=E4=B8=8E=E5=90=AF=E5=81=9C=E6=81=A2=E5=A4=8D?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../experiments/cloud-acceptance.md | 38 +++ .../experiments/evidence/grafana-plan.json | 14 + .../preview-browser-config-failed.json | 26 ++ .../preview-browser-config-timeout.json | 27 ++ .../evidence/preview-browser-disabled.json | 25 ++ .../evidence/preview-browser-fault.json | 270 ++++++++++++++++++ .../evidence/preview-browser-off.json | 19 ++ .../evidence/preview-browser-on.json | 245 ++++++++++++++++ .../evidence/preview-config-latency.json | 12 + .../evidence/preview-metric-parity.json | 6 + .../evidence/preview-metric-query.json | 72 +++++ .../experiments/evidence/preview-query.json | 48 ++++ .../evidence/preview-restored.json | 7 + .../experiments/evidence/preview-ui.json | 26 ++ .../experiments/preview-browser.mjs | 176 ++++++++++++ .../experiments/preview-control.py | 185 ++++++++++++ .../experiments/preview-metric-parity.mjs | 96 +++++++ .../experiments/preview-metric-query.py | 92 ++++++ .../experiments/preview-query.py | 189 ++++++++++++ .../experiments/preview-ui.mjs | 68 +++++ tasks/observability-foundation/packet.md | 22 +- .../observability-foundation/verification.md | 6 +- 22 files changed, 1660 insertions(+), 9 deletions(-) create mode 100644 tasks/observability-foundation/experiments/evidence/grafana-plan.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-browser-config-failed.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-browser-config-timeout.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-browser-disabled.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-browser-fault.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-browser-off.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-browser-on.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-config-latency.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-metric-parity.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-metric-query.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-query.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-restored.json create mode 100644 tasks/observability-foundation/experiments/evidence/preview-ui.json create mode 100644 tasks/observability-foundation/experiments/preview-browser.mjs create mode 100644 tasks/observability-foundation/experiments/preview-control.py create mode 100644 tasks/observability-foundation/experiments/preview-metric-parity.mjs create mode 100644 tasks/observability-foundation/experiments/preview-metric-query.py create mode 100644 tasks/observability-foundation/experiments/preview-query.py create mode 100644 tasks/observability-foundation/experiments/preview-ui.mjs diff --git a/tasks/observability-foundation/experiments/cloud-acceptance.md b/tasks/observability-foundation/experiments/cloud-acceptance.md index e592059..325a076 100644 --- a/tasks/observability-foundation/experiments/cloud-acceptance.md +++ b/tasks/observability-foundation/experiments/cloud-acceptance.md @@ -9,3 +9,41 @@ 实际首次TLS往返超过一秒,原固定一秒超时导致错误丢弃;改用标准可配置timeout后成功。Trace/Metrics返回200,Logs返回204;仅凭这些响应不声明存储通过,后续API读回才形成结论。服务端relay兼容200/204并仍只处理标准protobuf,实际浏览器relay的逐值ID验证见 [client-real-relay-result.json](client-real-relay-result.json)。浏览器该轮使用合成上游,不能把组合证据写成浏览器直连本stack的实测。 重跑用 `pdm run python tasks/observability-foundation/experiments/cloud-probe.py` 发送新合成运行,再运行同目录 `cloud-query.py` 查询对应运行。发送脚本只显式启用自己的子进程,保持本地 `.env` 的应用开关false。样本API导出不等于全历史、跨保留窗口迁移或厂商备份/恢复已验。 + +## PR 预览中的真实 Job 链路(2026-10-03) + +本轮使用 PR #124 的 Heroku Core 与隔离 Neon/PostgREST,临时启用 PostgreSQL 日志和 +私密 OTLP 出口,部署身份为 `19c771e2-e728-4299-b81a-9d7c387b207b`。Chromium 加载 +client-web 的真实业务模块,通过 PostgREST 提交最多处理一条记录的 lexical maintenance +Job,由远端 Core 领取执行;这不是付费模型调用,也不是在 Pages UI 中提交 Job。 +Pages 的实际预览随后独立显示成功 Job 和失败 Job 的 PostgreSQL 日志。 + +第一次和第二次运行暴露了实际问题:浏览器 1.5 秒的共享配置读取上限取消了合法请求。 +独立 HTTP 读取耗时 2.376 秒;业务成功和错误日志不受影响,但遥测未初始化,因此这两轮 +不算链路通过。客户端提交 `b64e5eb` 将读取上限调到 10 秒,直发 exporter 为 10 秒,relay +exporter 为 35 秒,以容纳服务端最多 32 秒的转发预算和网络余量。处理器多留 5 秒;调用方 +flush/shutdown 等待仍为 1.5 秒,不承诺页面退出排空。 + +修复后的 Job #6 成功,#7 按预期因无效参数失败并留下 1 条 PG 日志。 +`preview-query.json` 通过独立 Viewer API 读回两个服务的三信号:提交 Trace +`bddd7663505c46d94e35c215f4b7fff2`,执行 Trace `61e1f23b243c2eaf36856cd4be07da1a`, +执行 Span Link 精确指向提交 Span;`job.submitted`、`job.started`、`job.closed` 齐全, +浏览器提交与 Core 执行的时延指标均存在。初次读回发现浏览器的 `inkcre_operation` 和 Core 的 `operation` 标签不同; +JS 默认秒分桶也不能细分短请求。最终客户端 `6026b2d` 将时延指标标签与显式秒分桶 +对齐 Core,日志属性不变。历史 `preview-query.json` 保留当时两种标签的真实读回, +最终指标另由 `preview-metric-query.py` 严格核对统一标签、全部桶边界及 count/sum。 + +`preview-ui.json` 记录实际 Pages 预览中的成功状态与错误日志展示。浏览器 SDK 的部分 +fetch 出口在收到 200 后出现 `net::ERR_ABORTED` 事件;已查 SDK 的成功分支不消费 +响应体,但未据此断言这些事件的唯一原因。三信号是否成功以 +Cloud 独立读回为依据,不把网络事件数量当作精确导出或丢失账本。 + +账号 frontend settings 当前返回 `plan=free-trial`,见 `grafana-plan.json`。官方 Free +页面列出 10,000 活跃指标序列、每月各 50 GB 日志/Trace 及 14 天保留,但这不证明本账号 +已经结束试用,也不等于实际限额行为已验。本轮没有升级套餐或开启额外托管功能,持续 +生产 opt-in 仍需确认稳定 Free 状态及平台生命周期约束。 + +证据文件均位于本目录的 `evidence/`。真实凭据只经本地 Keychain/dotenv 和 Heroku API +传输;恢复快照在被忽略的 `runtime/`,不进入归档。 + +故障与恢复矩阵也已执行:共享配置请求始终不返回时,初始化在 10015.6ms 后停用遥测,Job #8 完成,#9 失败并保留 PG 日志;将预览 OTLP 出口切到本机拒连端口后,三信号 relay 返回 502,Job #10 仍完成,#11 的 PG 日志保留。仅关闭遥测后,#12 完成且 carrier 为 NULL,#13 的 PG 日志保留,浏览器无 OTLP 请求。最终指标严格读回 count=1、sum=0.0263s,全部显式桶边界与 Core 一致。 diff --git a/tasks/observability-foundation/experiments/evidence/grafana-plan.json b/tasks/observability-foundation/experiments/evidence/grafana-plan.json new file mode 100644 index 0000000..5022743 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/grafana-plan.json @@ -0,0 +1,14 @@ +{ + "observed_at": "2026-10-03", + "plan": "free-trial", + "stack_id": "1853809", + "scope": "Viewer frontend settings; not a billing-account contract", + "free_documented_limits": { + "active_metrics": 10000, + "monthly_logs_gb": 50, + "monthly_traces_gb": 50, + "retention_days": 14 + }, + "source": "https://grafana.com/products/cloud/free-tier/", + "production_opt_in_admitted": false +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-browser-config-failed.json b/tasks/observability-foundation/experiments/evidence/preview-browser-config-failed.json new file mode 100644 index 0000000..6b1f1f4 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-browser-config-failed.json @@ -0,0 +1,26 @@ +{ + "startedAt": "2026-10-03T13:00:54.492Z", + "enabled": true, + "browser": "149.0.7827.55", + "core": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com", + "pg": "https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/", + "relay": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com/telemetry", + "exports": [], + "errors": [ + "[Telemetry] Shared observability configuration is unavailable." + ], + "clientSource": "df9075551c6d05ca86dd5476043ab65fec1d9e1a", + "business": { + "id": 2, + "status": "finished", + "submissionTraceparent": null, + "carrierPreserved": true, + "invalidJob": { + "id": 3, + "status": "failed", + "pgLogs": 1 + } + }, + "finishedAt": "2026-10-03T13:03:09.579Z", + "passed": false +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-browser-config-timeout.json b/tasks/observability-foundation/experiments/evidence/preview-browser-config-timeout.json new file mode 100644 index 0000000..f7b0389 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-browser-config-timeout.json @@ -0,0 +1,27 @@ +{ + "startedAt": "2026-10-03T13:11:12.313Z", + "enabled": true, + "browser": "149.0.7827.55", + "core": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com", + "pg": "https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/", + "relay": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com/telemetry", + "exports": [], + "errors": [ + "[Telemetry] Shared observability configuration is unavailable." + ], + "clientSource": "b64e5eb5d6563b039e49530ea18296d10e519902", + "business": { + "initializationMs": 10015.599999904633, + "id": 8, + "status": "finished", + "submissionTraceparent": null, + "carrierPreserved": true, + "invalidJob": { + "id": 9, + "status": "failed", + "pgLogs": 1 + } + }, + "finishedAt": "2026-10-03T13:12:39.771Z", + "passed": true +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-browser-disabled.json b/tasks/observability-foundation/experiments/evidence/preview-browser-disabled.json new file mode 100644 index 0000000..e14dd25 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-browser-disabled.json @@ -0,0 +1,25 @@ +{ + "startedAt": "2026-10-03T13:19:16.078Z", + "enabled": false, + "browser": "149.0.7827.55", + "core": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com", + "pg": "https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/", + "relay": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com/telemetry", + "exports": [], + "errors": [], + "clientSource": "6026b2d9248b7faf7ebced7245565626f47008e7", + "business": { + "initializationMs": 4.699999809265137, + "id": 12, + "status": "finished", + "submissionTraceparent": null, + "carrierPreserved": true, + "invalidJob": { + "id": 13, + "status": "failed", + "pgLogs": 1 + } + }, + "finishedAt": "2026-10-03T13:20:23.490Z", + "passed": true +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-browser-fault.json b/tasks/observability-foundation/experiments/evidence/preview-browser-fault.json new file mode 100644 index 0000000..50d5e99 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-browser-fault.json @@ -0,0 +1,270 @@ +{ + "startedAt": "2026-10-03T13:12:13.289Z", + "enabled": true, + "browser": "149.0.7827.55", + "core": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com", + "pg": "https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/", + "relay": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com/telemetry", + "exports": [ + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "metrics", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "metrics", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "metrics", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "metrics", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "metrics", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "traces", + "status": 502 + }, + { + "signal": "metrics", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "logs", + "status": 502 + }, + { + "signal": "metrics", + "status": 502 + } + ], + "errors": [], + "clientSource": "b64e5eb5d6563b039e49530ea18296d10e519902", + "business": { + "initializationMs": 2647, + "id": 10, + "status": "finished", + "submissionTraceparent": "00-ca81fa3b0cd66bcd311dc2cae674bc92-01bed419ed5d75c4-01", + "carrierPreserved": true, + "invalidJob": { + "id": 11, + "status": "failed", + "pgLogs": 1 + } + }, + "finishedAt": "2026-10-03T13:14:16.093Z", + "passed": true +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-browser-off.json b/tasks/observability-foundation/experiments/evidence/preview-browser-off.json new file mode 100644 index 0000000..eec32a0 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-browser-off.json @@ -0,0 +1,19 @@ +{ + "startedAt": "2026-10-03T12:56:37.723Z", + "enabled": false, + "browser": "149.0.7827.55", + "core": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com", + "pg": "https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/", + "relay": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com/telemetry", + "exports": [], + "errors": [], + "clientSource": "df9075551c6d05ca86dd5476043ab65fec1d9e1a", + "business": { + "id": 1, + "status": "finished", + "submissionTraceparent": null, + "carrierPreserved": true + }, + "finishedAt": "2026-10-03T12:57:16.597Z", + "passed": true +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-browser-on.json b/tasks/observability-foundation/experiments/evidence/preview-browser-on.json new file mode 100644 index 0000000..7534a66 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-browser-on.json @@ -0,0 +1,245 @@ +{ + "startedAt": "2026-10-03T13:06:49.836Z", + "enabled": true, + "browser": "149.0.7827.55", + "core": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com", + "pg": "https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/", + "relay": "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com/telemetry", + "exports": [ + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "metrics", + "status": 200 + }, + { + "signal": "metrics", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "traces", + "status": 200 + }, + { + "signal": "metrics", + "status": 200 + }, + { + "signal": "metrics", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "metrics", + "status": 200 + }, + { + "signal": "metrics", + "failure": "net::ERR_ABORTED" + }, + { + "signal": "metrics", + "status": 200 + }, + { + "signal": "metrics", + "failure": "net::ERR_ABORTED" + } + ], + "errors": [], + "clientSource": "b64e5eb5d6563b039e49530ea18296d10e519902", + "business": { + "id": 6, + "status": "finished", + "submissionTraceparent": "00-bddd7663505c46d94e35c215f4b7fff2-e7bae644977cddd7-01", + "carrierPreserved": true, + "invalidJob": { + "id": 7, + "status": "failed", + "pgLogs": 1 + } + }, + "finishedAt": "2026-10-03T13:09:09.275Z", + "passed": true +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-config-latency.json b/tasks/observability-foundation/experiments/evidence/preview-config-latency.json new file mode 100644 index 0000000..6529232 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-config-latency.json @@ -0,0 +1,12 @@ +{ + "observed_at": "2026-10-03", + "preview": "core-py PR #124", + "browser": "Chromium 149.0.7827.55", + "browser_config_timeout_ms": 1500, + "browser_failure": "net::ERR_ABORTED", + "independent_http_config_read_seconds": 2.376, + "shared_schema": "inkcre.observability.v1", + "shared_configuration_valid": true, + "business_job_finished": true, + "postgresql_error_job_log_count": 1 +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-metric-parity.json b/tasks/observability-foundation/experiments/evidence/preview-metric-parity.json new file mode 100644 index 0000000..88d5ea1 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-metric-parity.json @@ -0,0 +1,6 @@ +{ + "version": "preview-metric-parity-20261003", + "source": "6026b2d9248b7faf7ebced7245565626f47008e7", + "traceparent": "00-ea75aa4e374c1fb882846f289b7709c4-60df2209cc411b45-01", + "finishedAt": "2026-10-03T13:18:20.240Z" +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-metric-query.json b/tasks/observability-foundation/experiments/evidence/preview-metric-query.json new file mode 100644 index 0000000..2295eb4 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-metric-query.json @@ -0,0 +1,72 @@ +{ + "verified": true, + "source": "6026b2d9248b7faf7ebced7245565626f47008e7", + "labels": { + "operation": "job.submit", + "outcome": "success" + }, + "count": 1.0, + "sum_seconds": 0.026299999713897706, + "buckets": [ + { + "le": "0.005", + "count": 0.0 + }, + { + "le": "0.01", + "count": 0.0 + }, + { + "le": "0.025", + "count": 0.0 + }, + { + "le": "0.05", + "count": 1.0 + }, + { + "le": "0.1", + "count": 1.0 + }, + { + "le": "0.25", + "count": 1.0 + }, + { + "le": "0.5", + "count": 1.0 + }, + { + "le": "1.0", + "count": 1.0 + }, + { + "le": "2.5", + "count": 1.0 + }, + { + "le": "5.0", + "count": 1.0 + }, + { + "le": "10.0", + "count": 1.0 + }, + { + "le": "30.0", + "count": 1.0 + }, + { + "le": "60.0", + "count": 1.0 + }, + { + "le": "300.0", + "count": 1.0 + }, + { + "le": "inf", + "count": 1.0 + } + ] +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-query.json b/tasks/observability-foundation/experiments/evidence/preview-query.json new file mode 100644 index 0000000..193f325 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-query.json @@ -0,0 +1,48 @@ +{ + "deployment_id": "19c771e2-e728-4299-b81a-9d7c387b207b", + "job_id": 6, + "source": "b64e5eb5d6563b039e49530ea18296d10e519902", + "signals": { + "traces": { + "verified": true, + "submit_trace_id": "bddd7663505c46d94e35c215f4b7fff2", + "execution_trace_id": "61e1f23b243c2eaf36856cd4be07da1a", + "span_count": 2, + "services": [ + "core-py", + "inkcre.client-web" + ] + }, + "logs": { + "verified": true, + "events": [ + "job.closed", + "job.started", + "job.submitted" + ] + }, + "metrics": { + "verified": true, + "observed": [ + [ + "core-py", + "http.server" + ], + [ + "core-py", + "job.execute" + ], + [ + "inkcre.client-web", + "job.submit" + ], + [ + "inkcre.client-web", + "postgrest.http" + ] + ], + "series_count": 6 + } + }, + "verified": true +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-restored.json b/tasks/observability-foundation/experiments/evidence/preview-restored.json new file mode 100644 index 0000000..389e263 --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-restored.json @@ -0,0 +1,7 @@ +{ + "restored": true, + "telemetry_enabled": "false", + "logging_backend": "none", + "shared_config_restored": true, + "private_export_keys_removed": true +} diff --git a/tasks/observability-foundation/experiments/evidence/preview-ui.json b/tasks/observability-foundation/experiments/evidence/preview-ui.json new file mode 100644 index 0000000..5ef709b --- /dev/null +++ b/tasks/observability-foundation/experiments/evidence/preview-ui.json @@ -0,0 +1,26 @@ +{ + "origin": "https://preview-client-web-pr-125.inkcre-client-web.pages.dev", + "jobId": 6, + "errorJobId": 7, + "requests": [ + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "logs", + "status": 200 + }, + { + "signal": "traces", + "status": 200 + } + ], + "verified": true, + "finishedVisible": true, + "pgLogVisible": true +} diff --git a/tasks/observability-foundation/experiments/preview-browser.mjs b/tasks/observability-foundation/experiments/preview-browser.mjs new file mode 100644 index 0000000..fe8754a --- /dev/null +++ b/tasks/observability-foundation/experiments/preview-browser.mjs @@ -0,0 +1,176 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { writeFile } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { fileURLToPath } from 'node:url' +const client = fileURLToPath(new URL('../../../../client-web', import.meta.url)) +const require = createRequire(`${client}/package.json`) +const { chromium } = require('@playwright/test') +const { createServer } = await import( + `${client}/apps/client-web/node_modules/vite/dist/node/index.js` +) +const jwt = execFileSync( + 'security', + ['find-generic-password', '-s', 'inkcre/core-py/JWT_SECRET', '-w'], + { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] } +).trim() +const pg = 'https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/' +const core = 'https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com' +const relay = core + '/telemetry' +const enabled = !process.argv.includes('--off') +const configFault = process.argv.includes('--config-fault') +const label = process.env.INKCRE_PREVIEW_LABEL || (enabled ? 'on' : 'off') +const vite = await createServer({ + configFile: false, + root: `${client}/packages/core`, + server: { host: '127.0.0.1', port: 0, fs: { allow: [client] } }, + optimizeDeps: { + entries: [], + include: [ + 'vue', + 'pinia', + 'zod', + 'zod-class', + 'zod-config', + 'jose', + '@supabase/postgrest-js', + '@opentelemetry/api', + '@opentelemetry/core', + '@opentelemetry/resources', + '@opentelemetry/sdk-trace-web', + '@opentelemetry/exporter-trace-otlp-proto', + '@opentelemetry/sdk-logs', + '@opentelemetry/exporter-logs-otlp-proto', + '@opentelemetry/sdk-metrics', + '@opentelemetry/exporter-metrics-otlp-proto', + ], + }, +}) +vite.middlewares.use('/probe', (_req, res) => { + res.setHeader('Content-Type', 'text/html') + res.end('InKCre preview acceptance') +}) +await vite.listen() +const browser = await chromium.launch({ headless: true }) +const report = { + startedAt: new Date().toISOString(), + enabled, + browser: await browser.version(), + core, + pg, + relay, + exports: [], + errors: [], + clientSource: execFileSync('git', ['-C', client, 'rev-parse', 'HEAD'], { + encoding: 'utf8', + }).trim(), +} +try { + const page = await browser.newPage() + if (configFault) await page.route('**/configs?**', () => {}) + page.on('console', (message) => { + if (message.text().startsWith('[Telemetry]')) report.errors.push(message.text()) + }) + page.on('requestfinished', (request) => { + if (request.url().includes('/configs?')) + console.log(JSON.stringify({ configReadMs: request.timing().responseEnd, path: '/configs' })) + }) + page.on('requestfailed', (request) => { + if (request.url().includes('/configs?')) + console.log( + JSON.stringify({ + configReadFailure: request.failure()?.errorText, + duration: request.timing().responseEnd, + }) + ) + }) + page.on('response', (response) => { + if (response.url().includes('/configs?')) + console.log(JSON.stringify({ configReadStatus: response.status() })) + if (response.url().startsWith(relay + '/v1/')) + report.exports.push({ signal: response.url().split('/').at(-1), status: response.status() }) + }) + page.on('requestfailed', (request) => { + if (request.url().startsWith(relay + '/v1/')) + report.exports.push({ + signal: request.url().split('/').at(-1), + failure: request.failure()?.errorText, + }) + }) + await page.goto(`http://127.0.0.1:${vite.httpServer.address().port}/probe`) + report.business = await page.evaluate( + async ({ pg, jwt, relay, core, enabled, configFault }) => { + const t = await import('/src/obsrv/telemetry.ts') + const { configStore } = await import('/src/config/store.ts') + const { MetaConfigSchema } = await import('/src/config/schema.ts') + const { Job } = await import('/src/job/job.ts') + const { JobManager } = await import('/src/job/manager.ts') + const { signDatabaseToken } = await import('/src/auth/index.ts') + configStore.metaConfig = MetaConfigSchema.parse({ + INKCRE_PGREST_URL: pg, + INKCRE_JWT_SECRET: jwt, + INKCRE_PEER_ID: '00000000-0000-4000-8000-000000000097', + telemetry_enabled: enabled, + telemetry_peer_relay_url: relay, + }) + const initializationStarted = performance.now() + await t.initializeTelemetry(configStore.metaConfig, 'preview-acceptance') + const initializationMs = performance.now() - initializationStarted + const job = await JobManager.create( + 'core.feature_retrieval.lexical.maintain.v1', + { options: { max_records: 1, scan_page_size: 1, diagnostic_limit: 0 } }, + 60 + ) + let saved = job + for (let n = 0; n < 24 && ['pending', 'running'].includes(saved.status); n++) { + await new Promise((r) => setTimeout(r, 2500)) + saved = await Job.get(job.id) + } + const invalid = await JobManager.create( + 'core.feature_retrieval.lexical.maintain.v1', + { options: { max_records: 0 } }, + 60 + ) + let rejected = invalid + for (let n = 0; n < 24 && ['pending', 'running'].includes(rejected.status); n++) { + await new Promise((r) => setTimeout(r, 2500)) + rejected = await Job.get(invalid.id) + } + await new Promise((r) => setTimeout(r, 1000)) + const { DBAPIClient } = await import('/src/base/db-api.ts') + const logs = await new DBAPIClient('logs') + .from() + .select('id,trace_id') + .eq('trace_id', `job.${invalid.id}`) + .throwOnError() + await t.flushTelemetry() + if (enabled && !configFault) await new Promise((r) => setTimeout(r, 65000)) + await t.shutdownTelemetry() + return { + initializationMs, + id: job.id, + status: saved.status, + submissionTraceparent: job.submission_traceparent, + carrierPreserved: saved.submission_traceparent === job.submission_traceparent, + invalidJob: { id: invalid.id, status: rejected.status, pgLogs: logs.data.length }, + } + }, + { pg, jwt, relay, core, enabled, configFault } + ) + report.finishedAt = new Date().toISOString() + report.passed = + report.business.status === 'finished' && + report.business.carrierPreserved && + (enabled && !configFault + ? Boolean(report.business.submissionTraceparent) + : report.business.submissionTraceparent === null) +} finally { + await browser.close() + await vite.close() + await writeFile( + new URL(`./evidence/preview-browser-${label}.json`, import.meta.url), + JSON.stringify(report, null, 2) + '\n' + ) + console.log(JSON.stringify(report)) +} +assert(report.passed, 'Preview job must finish and preserve carrier') diff --git a/tasks/observability-foundation/experiments/preview-control.py b/tasks/observability-foundation/experiments/preview-control.py new file mode 100644 index 0000000..2b67a73 --- /dev/null +++ b/tasks/observability-foundation/experiments/preview-control.py @@ -0,0 +1,185 @@ +"""Bounded PR-124 acceptance; private values stay in memory or ignored restore file.""" + +import json +import os +from pathlib import Path +import subprocess +import sys +import time +import uuid + +from dotenv import dotenv_values +import httpx +import jwt + +HERE = Path(__file__).resolve().parent +APP = "inkcre-core-py-pr-124" +CORE = "https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com" +PG = "https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com" +PRIVATE = HERE / "runtime/preview-restore.json" +secret = subprocess.check_output( + ["security", "find-generic-password", "-s", "inkcre/core-py/JWT_SECRET", "-w"], text=True +).strip() +token = jwt.encode( + { + "role": "authenticated", + "iss": "inkcre-peer", + "aud": "inkcre-api", + "iat": int(time.time()), + "exp": int(time.time()) + 3600, + }, + secret, + algorithm="HS256", +) +headers = { + "Authorization": "Bearer " + token, + "Accept-Profile": "inkcre", + "Content-Profile": "inkcre", +} +local = dotenv_values(HERE.parents[2] / ".env") + + +def request(method, url, **kwargs): + response = httpx.request(method, url, timeout=60, **kwargs) + if response.status_code >= 400: + raise RuntimeError(f"HTTP {response.status_code} from {url.split('?')[0]}") + return response + + +def heroku(method, path, **kwargs): + credential = subprocess.check_output( + ["heroku", "auth:token"], text=True, stderr=subprocess.DEVNULL + ).strip() + return request( + method, + "https://api.heroku.com/apps/" + APP + path, + headers={ + "Authorization": "Bearer " + credential, + "Accept": "application/vnd.heroku+json; version=3", + }, + **kwargs, + ) + + +mode = sys.argv[1] +if mode == "inspect": + config = heroku("GET", "/config-vars").json() + shared = request( + "GET", + PG + "/configs?key=eq.inkcre.observability&select=key,schema,value", + headers=headers, + ).json() + types = request("GET", CORE + "/job-types", headers=headers).json() + print( + json.dumps( + { + "telemetry_enabled": config.get("OBSRV__TELEMETRY_ENABLED", "false"), + "logging_backend": config.get("OBSRV__LOGGING_BACKEND"), + "shared_identity_present": bool(shared), + "job_types": [x["id"] for x in types["job_types"]], + "otlp_keys_present": sorted(k for k in config if k.startswith("OTEL_")), + } + ) + ) +elif mode == "on": + assert not PRIVATE.exists(), "Restore snapshot already exists" + config = heroku("GET", "/config-vars").json() + update = {k: v for k, v in local.items() if k.startswith("OTEL_EXPORTER_OTLP") and v} + assert any(k.endswith("_ENDPOINT") for k in update) + update.update( + { + "OBSRV__TELEMETRY_ENABLED": "true", + "OBSRV__LOGGING_BACKEND": "postgresql", + "OTEL_PYTHON_SDK_INTERNAL_METRICS_ENABLED": "true", + } + ) + shared = request( + "GET", + PG + "/configs?key=eq.inkcre.observability&select=key,schema,value", + headers=headers, + ).json() + run_id = str(uuid.uuid4()) + PRIVATE.parent.mkdir(exist_ok=True) + fd = os.open(PRIVATE, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "w") as stream: + json.dump( + {"config": {k: config.get(k) for k in update}, "shared": shared, "run_id": run_id}, + stream, + ) + request( + "POST", + PG + "/configs?on_conflict=key", + headers={**headers, "Prefer": "resolution=merge-duplicates"}, + json={ + "key": "inkcre.observability", + "schema": "inkcre.observability.v1", + "value": { + "deployment_id": run_id, + "otlp_http_endpoints": None, + "diagnostics_url": None, + }, + }, + ) + heroku("PATCH", "/config-vars", json=update) + print( + json.dumps( + {"preview_enabled": True, "deployment_id": run_id, "app": APP, "restore_saved": True} + ) + ) +elif mode == "healthy": + assert PRIVATE.exists() + heroku( + "PATCH", + "/config-vars", + json={k: v for k, v in local.items() if k.startswith("OTEL_EXPORTER_OTLP") and v}, + ) + print(json.dumps({"preview_upstream_restored": True})) +elif mode == "disable": + heroku("PATCH", "/config-vars", json={"OBSRV__TELEMETRY_ENABLED": "false"}) + print(json.dumps({"preview_telemetry_enabled": False, "logging_backend": "postgresql"})) +elif mode == "fault": + assert PRIVATE.exists() + heroku( + "PATCH", + "/config-vars", + json={ + f"OTEL_EXPORTER_OTLP_{signal}_ENDPOINT": f"http://127.0.0.1:9/v1/{signal.lower()}" + for signal in ("TRACES", "LOGS", "METRICS") + }, + ) + print(json.dumps({"preview_upstream_fault": "loopback_refused"})) +elif mode == "off": + saved = json.loads(PRIVATE.read_text()) + heroku("PATCH", "/config-vars", json=saved["config"]) + if saved["shared"]: + request( + "POST", + PG + "/configs?on_conflict=key", + headers={**headers, "Prefer": "resolution=merge-duplicates"}, + json=saved["shared"][0], + ) + else: + request("DELETE", PG + "/configs?key=eq.inkcre.observability", headers=headers) + for attempt in range(6): + config = heroku("GET", "/config-vars").json() + if all(config.get(k) == v for k, v in saved["config"].items()): + break + time.sleep(2) + else: + raise RuntimeError("Preview configuration restore did not converge") + print( + json.dumps( + { + "restored": True, + "telemetry_enabled": config.get("OBSRV__TELEMETRY_ENABLED", "false"), + "private_export_keys_removed": all( + not config.get(k) + for k, v in saved["config"].items() + if k.startswith("OTEL_EXPORTER_OTLP") and v is None + ), + } + ) + ) +elif mode == "ready": + r = httpx.get(CORE + "/readyz", timeout=60) + print(json.dumps({"ready_status": r.status_code})) diff --git a/tasks/observability-foundation/experiments/preview-metric-parity.mjs b/tasks/observability-foundation/experiments/preview-metric-parity.mjs new file mode 100644 index 0000000..b28728e --- /dev/null +++ b/tasks/observability-foundation/experiments/preview-metric-parity.mjs @@ -0,0 +1,96 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { writeFile } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { fileURLToPath } from 'node:url' +const client = fileURLToPath(new URL('../../../../client-web', import.meta.url)) +const require = createRequire(`${client}/package.json`) +const { chromium } = require('@playwright/test') +const { createServer } = await import( + `${client}/apps/client-web/node_modules/vite/dist/node/index.js` +) +const jwt = execFileSync( + 'security', + ['find-generic-password', '-s', 'inkcre/core-py/JWT_SECRET', '-w'], + { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] } +).trim() +const pg = 'https://inkcre-postgrest-pr-124-c8a3abf570a2.herokuapp.com/' +const core = 'https://inkcre-core-py-pr-124-17b00650cc5c.herokuapp.com' +const relay = core + '/telemetry' +const vite = await createServer({ + configFile: false, + root: `${client}/packages/core`, + server: { host: '127.0.0.1', port: 0, fs: { allow: [client] } }, + optimizeDeps: { + entries: [], + include: [ + 'vue', + 'pinia', + 'zod', + 'zod-class', + 'zod-config', + 'jose', + '@supabase/postgrest-js', + '@opentelemetry/api', + '@opentelemetry/core', + '@opentelemetry/resources', + '@opentelemetry/sdk-trace-web', + '@opentelemetry/exporter-trace-otlp-proto', + '@opentelemetry/sdk-logs', + '@opentelemetry/exporter-logs-otlp-proto', + '@opentelemetry/sdk-metrics', + '@opentelemetry/exporter-metrics-otlp-proto', + ], + }, +}) +vite.middlewares.use('/probe', (_req, res) => { + res.setHeader('Content-Type', 'text/html') + res.end('InKCre metric parity acceptance') +}) +await vite.listen() +const browser = await chromium.launch({ headless: true }) +const version = 'preview-metric-parity-20261003' +try { + const page = await browser.newPage() + await page.goto(`http://127.0.0.1:${vite.httpServer.address().port}/probe`) + const result = await page.evaluate( + async ({ pg, jwt, relay, version }) => { + const t = await import('/src/obsrv/telemetry.ts') + const { MetaConfigSchema } = await import('/src/config/schema.ts') + await t.initializeTelemetry( + MetaConfigSchema.parse({ + INKCRE_PGREST_URL: pg, + INKCRE_JWT_SECRET: jwt, + INKCRE_PEER_ID: '00000000-0000-4000-8000-000000000095', + telemetry_enabled: true, + telemetry_peer_relay_url: relay, + }), + version + ) + const carrier = await t.observeOperation('job.submit', async (active) => { + await new Promise((r) => setTimeout(r, 25)) + return t.captureSubmission(active) + }) + await t.flushTelemetry() + await new Promise((r) => setTimeout(r, 18000)) + await t.shutdownTelemetry() + return { traceparent: carrier.submission_traceparent } + }, + { pg, jwt, relay, version } + ) + const report = { + version, + source: execFileSync('git', ['-C', client, 'rev-parse', 'HEAD'], { encoding: 'utf8' }).trim(), + ...result, + finishedAt: new Date().toISOString(), + } + await writeFile( + new URL('./evidence/preview-metric-parity.json', import.meta.url), + JSON.stringify(report, null, 2) + '\n' + ) + console.log(JSON.stringify(report)) + assert(result.traceparent) +} finally { + await browser.close() + await vite.close() +} diff --git a/tasks/observability-foundation/experiments/preview-metric-query.py b/tasks/observability-foundation/experiments/preview-metric-query.py new file mode 100644 index 0000000..bb03fd2 --- /dev/null +++ b/tasks/observability-foundation/experiments/preview-metric-query.py @@ -0,0 +1,92 @@ +"""Assert Cloud-stored browser histogram interoperates with Core's published stream.""" + +from datetime import datetime +import json +from pathlib import Path + +from dotenv import dotenv_values +import httpx + +HERE = Path(__file__).resolve().parent +config = dotenv_values(HERE.parents[2] / ".env") +probe = json.loads((HERE / "evidence/preview-metric-parity.json").read_text()) +with httpx.Client( + base_url=config["GRAFANA_URL"], + headers={"Authorization": "Bearer " + config["GRAFANA_SERVICE_ACCOUNT_TOKEN"]}, + timeout=30, +) as client: + catalog = client.get("/api/datasources") + assert catalog.status_code == 200 + uid = next( + x["uid"] + for x in catalog.json() + if x["type"] == "prometheus" and x["name"].endswith("-prom") + ) + response = client.get( + "/api/datasources/proxy/uid/" + uid + "/api/v1/query", + params={ + "query": ( + '{__name__=~"inkcre_operation_duration_seconds_(bucket|count|sum)",' + 'service_version="' + ) + + probe["version"] + + '"}', + "time": int( + datetime.fromisoformat(probe["finishedAt"].replace("Z", "+00:00")).timestamp() + ) + + 10, + }, + ) + assert response.status_code == 200 + series = response.json().get("data", {}).get("result", []) + assert series, "Browser histogram absent" + assert len({x["metric"]["service_instance_id"] for x in series}) == 1 + assert all( + x["metric"].get("operation") == "job.submit" + and x["metric"].get("outcome") == "success" + and "inkcre_operation" not in x["metric"] + for x in series + ), "Cross-Peer metric labels differ" + buckets = sorted( + (float(x["metric"]["le"]), float(x["value"][1])) + for x in series + if x["metric"]["__name__"].endswith("_bucket") + ) + expected = [ + 0.005, + 0.01, + 0.025, + 0.05, + 0.1, + 0.25, + 0.5, + 1, + 2.5, + 5, + 10, + 30, + 60, + 300, + float("inf"), + ] + assert [b for b, _ in buckets] == expected, "Cross-Peer histogram boundaries differ" + assert all(a[1] <= b[1] for a, b in zip(buckets, buckets[1:])) and buckets[-1][1] == 1 + count = next( + float(x["value"][1]) for x in series if x["metric"]["__name__"].endswith("_count") + ) + total = next( + float(x["value"][1]) for x in series if x["metric"]["__name__"].endswith("_sum") + ) + assert count == 1 and total > 0 +report = { + "verified": True, + "source": probe["source"], + "labels": {"operation": "job.submit", "outcome": "success"}, + "count": count, + "sum_seconds": total, + "buckets": [{"le": str(b), "count": c} for b, c in buckets], +} +(HERE / "evidence/preview-metric-query.json").write_text( + json.dumps(report, indent=2) + "\n" +) +print(json.dumps(report)) diff --git a/tasks/observability-foundation/experiments/preview-query.py b/tasks/observability-foundation/experiments/preview-query.py new file mode 100644 index 0000000..21c3ca1 --- /dev/null +++ b/tasks/observability-foundation/experiments/preview-query.py @@ -0,0 +1,189 @@ +"""Independently read bounded preview Job evidence from Grafana datasource APIs.""" + +import base64 +from datetime import datetime +import json +from pathlib import Path +import re +import time + +from dotenv import dotenv_values +import httpx + +HERE = Path(__file__).resolve().parent +config = dotenv_values(HERE.parents[2] / ".env") +source = json.loads((HERE / "evidence/preview-browser-on.json").read_text()) +deployment = json.loads((HERE / "runtime/preview-restore.json").read_text())["run_id"] +job = source["business"] +traceparent = job["submissionTraceparent"].split("-") +start = ( + int(datetime.fromisoformat(source["startedAt"].replace("Z", "+00:00")).timestamp()) - 10 +) +end = int(time.time()) +report = { + "deployment_id": deployment, + "job_id": job["id"], + "source": source["clientSource"], + "signals": {}, +} + + +def ident(value): + return ( + value + if re.fullmatch(r"[0-9a-f]{16}|[0-9a-f]{32}", value) + else base64.b64decode(value).hex() + ) + + +def attrs(value): + return { + item["key"]: next(iter(item["value"].values())) for item in value.get("attributes", []) + } + + +def walk(value): + if isinstance(value, dict): + yield value + for child in value.values(): + yield from walk(child) + elif isinstance(value, list): + for child in value: + yield from walk(child) + + +with httpx.Client( + base_url=config["GRAFANA_URL"], + headers={"Authorization": "Bearer " + config["GRAFANA_SERVICE_ACCOUNT_TOKEN"]}, + timeout=30, +) as client: + + def get(path, **kwargs): + r = client.get(path, **kwargs) + if r.status_code != 200: + raise RuntimeError(f"Grafana status {r.status_code}, path {path}") + return r.json() + + catalog = get("/api/datasources") + ds = { + kind: "/api/datasources/proxy/uid/" + + next(x["uid"] for x in catalog if x["type"] == kind and x["name"].endswith(suffix)) + for kind, suffix in [("tempo", "-traces"), ("loki", "-logs"), ("prometheus", "-prom")] + } + search = get( + ds["tempo"] + "/api/search", + params={ + "q": ( + f'{{ resource.inkcre.deployment.id = "{deployment}" ' + f"&& span.inkcre.job.id = {job['id']} }}" + ), + "start": start, + "end": end, + "limit": 20, + }, + ) + ids = {traceparent[1]} | {x["traceID"] for x in search.get("traces", [])} + spans = [] + resources = [] + for trace_id in sorted(ids): + r = client.get( + ds["tempo"] + "/api/traces/" + trace_id, headers={"Accept": "application/json"} + ) + if r.status_code != 200: + continue + for item in walk(r.json()): + if "spanId" in item and "name" in item: + spans.append(item) + if "service.instance.id" in attrs(item): + resources.append(attrs(item)) + submission = next( + ( + s for s in spans if s["name"] == "job.submit" and ident(s["spanId"]) == traceparent[2] + ), + None, + ) + execution = next( + ( + s + for s in spans + if s["name"] == "job.execute" and int(attrs(s).get("inkcre.job.id", -1)) == job["id"] + ), + None, + ) + trace_ok = bool( + submission + and execution + and ident(execution["traceId"]) != traceparent[1] + and any( + ident(link["traceId"]) == traceparent[1] and ident(link["spanId"]) == traceparent[2] + for link in execution.get("links", []) + ) + ) + report["signals"]["traces"] = { + "verified": trace_ok, + "submit_trace_id": traceparent[1], + "execution_trace_id": ident(execution["traceId"]) if execution else None, + "span_count": len(spans), + "services": sorted({r.get("service.name", "") for r in resources}), + } + logs = get( + ds["loki"] + "/loki/api/v1/query_range", + params={ + "query": ( + '{service_name=~"core-py|inkcre.client-web"} ' + f'| inkcre_deployment_id="{deployment}" | inkcre_job_id="{job["id"]}"' + ), + "start": str(start * 10**9), + "end": str(end * 10**9), + "limit": 50, + }, + ) + events = [ + v[1] + for stream in logs.get("data", {}).get("result", []) + for v in stream.get("values", []) + ] + report["signals"]["logs"] = { + "verified": {"job.submitted", "job.started", "job.closed"}.issubset(events), + "events": events, + } + metric_rows = [] + for resource in resources: + result = get( + ds["prometheus"] + "/api/v1/query", + params={ + "query": ( + '{__name__=~"inkcre_operation_duration_seconds_count|' + 'inkcre_operation_count_total",service_instance_id="' + ) + + resource["service.instance.id"] + + '"}', + "time": min( + end, + int( + datetime.fromisoformat(source["finishedAt"].replace("Z", "+00:00")).timestamp() + ) + + 30, + ), + }, + ) + metric_rows.extend(result.get("data", {}).get("result", [])) + observed = { + ( + x["metric"].get("service_name"), + x["metric"].get("operation", x["metric"].get("inkcre_operation")), + ) + for x in metric_rows + if float(x["value"][1]) >= 1 + } + report["signals"]["metrics"] = { + "verified": {("inkcre.client-web", "job.submit"), ("core-py", "job.execute")}.issubset( + observed + ), + "observed": sorted(observed), + "series_count": len(metric_rows), + } +report["verified"] = all(x["verified"] for x in report["signals"].values()) +(HERE / "evidence/preview-query.json").write_text(json.dumps(report, indent=2) + "\n") +print(json.dumps(report, indent=2)) +assert report["verified"], "Preview Cloud readback is incomplete" diff --git a/tasks/observability-foundation/experiments/preview-ui.mjs b/tasks/observability-foundation/experiments/preview-ui.mjs new file mode 100644 index 0000000..828e9cf --- /dev/null +++ b/tasks/observability-foundation/experiments/preview-ui.mjs @@ -0,0 +1,68 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFile, writeFile } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { fileURLToPath } from 'node:url' +const client = fileURLToPath(new URL('../../../../client-web', import.meta.url)) +const require = createRequire(`${client}/package.json`) +const { chromium } = require('@playwright/test') +const jwt = execFileSync( + 'security', + ['find-generic-password', '-s', 'inkcre/core-py/JWT_SECRET', '-w'], + { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] } +).trim() +const source = JSON.parse( + await readFile(new URL('./evidence/preview-browser-on.json', import.meta.url)) +) +const origin = 'https://preview-client-web-pr-125.inkcre-client-web.pages.dev' +const browser = await chromium.launch({ headless: true }) +const report = { + origin, + jobId: source.business.id, + errorJobId: source.business.invalidJob.id, + requests: [], + verified: false, +} +try { + const page = await browser.newPage() + await page.addInitScript( + ({ origin, jwt, pg, relay }) => { + if (location.origin === origin) + localStorage.setItem( + 'inkcre_app_config', + JSON.stringify({ + INKCRE_PGREST_URL: pg, + INKCRE_JWT_SECRET: jwt, + INKCRE_PEER_ID: '00000000-0000-4000-8000-000000000096', + telemetry_enabled: true, + telemetry_peer_relay_url: relay, + }) + ) + }, + { origin, jwt, pg: source.pg, relay: source.relay } + ) + page.on('response', (response) => { + if (response.url().includes('/telemetry/v1/')) + report.requests.push({ signal: response.url().split('/').at(-1), status: response.status() }) + }) + await page.goto(origin + '/jobs/' + report.jobId) + await page.locator('.status--finished').waitFor({ timeout: 60000 }) + report.finishedVisible = true + await page.goto(origin + '/jobs/' + report.errorJobId) + await page.locator('.status--failed').waitFor({ timeout: 60000 }) + await page + .getByText('Persisted Job parameters are invalid', { exact: false }) + .first() + .waitFor({ timeout: 30000 }) + report.pgLogVisible = true + report.verified = true + await page.waitForTimeout(10000) +} finally { + await browser.close() + await writeFile( + new URL('./evidence/preview-ui.json', import.meta.url), + JSON.stringify(report, null, 2) + '\n' + ) + console.log(JSON.stringify(report)) +} +assert(report.verified) diff --git a/tasks/observability-foundation/packet.md b/tasks/observability-foundation/packet.md index c40a916..45866d9 100644 --- a/tasks/observability-foundation/packet.md +++ b/tasks/observability-foundation/packet.md @@ -1,17 +1,17 @@ # InKCre 可观测性基建 -建立适合多 Peer、serverless 与 scale-to-0 的可选观测能力。Sir 已选定 Grafana Cloud Free,要求 vendor-agnostic、默认维持 PostgreSQL 日志,并已授权实施、创建分支、提交、推送与 draft PR。没有合并、生产发布或付费服务授权。 +建立适合多 Peer、serverless 与 scale-to-0 的可选观测能力。Sir 已选定 Grafana Cloud Free,要求 vendor-agnostic、默认维持 PostgreSQL 日志,并已授权实施、创建分支、提交、推送与 draft PR。Sir 已进一步授权推进既定验收与 PR 合并顺序,包括正常 Release PR 所触发的交付;没有付费服务或持续生产遥测启用授权。 ## 当前交付 -首批实现已进入集成收尾:Python 与 client-web 使用标准 OTel/OTLP,本地开关缺省关闭,PG writer 与 Job 查询保留。Hub 契约已发布到分支;两 Spoke 的共享引用各自独立提交。两个 Spoke 的源码均已提交、推送并发布 draft PR;任务包与实验归档作为独立提交收尾。 +首批实现正在完成预览验收与按依赖合并:Python 与 client-web 使用标准 OTel/OTLP,本地开关缺省关闭,PG writer 与 Job 查询保留。Hub #33 已合并;两 Spoke 各自独立更新共享引用至 `7916312`,#124/#125 已退出 draft。core 功能合并后先走既有 Release PR 和 stable 准入,再同步客户端类型与合并。 | 对象 | 当前状态 | | --- | --- | -| Hub | `codex/observability-contract`,`a603b036fd41dae5a471dea24cd7609bb4927bd5`,[draft PR #33](https://github.com/InKCre/docs/pull/33) | -| core-py | [draft PR #124](https://github.com/InKCre/core-py/pull/124),源码 `0eba26a`,共享引用提交 `f0397e7`,基于 `fa23a1c`;默认关闭、三信号、HTTP/Peer/Job/Cron/AI/Agent/来源事件、正式迁移及受控 protobuf relay | -| client-web | [draft PR #125](https://github.com/InKCre/client-web/pull/125),源码 `b81554d`,共享引用提交 `b1072a9`,基于 `d4c6437`;连接本地开关、三信号、Peer/Job carrier/Link、可选诊断链接与认证 relay | -| Grafana Cloud | Sir 已创建 stack 并配置写入与 Viewer 凭据。三类合成数据已由 Viewer API 独立读回;Trace/Span/Link、AI unknown/zero、日志和指标均与源输入一致 | +| Hub | `main`,`7916312c12cfaf2ee396d7dc1afffee67d4f0721`,[已合并 PR #33](https://github.com/InKCre/docs/pull/33) | +| core-py | [PR #124](https://github.com/InKCre/core-py/pull/124),源码 `0eba26a`,共享引用提交 `4e4b10e`,基于 `fa23a1c`;默认关闭、三信号、HTTP/Peer/Job/Cron/AI/Agent/来源事件、正式迁移及受控 protobuf relay | +| client-web | [PR #125](https://github.com/InKCre/client-web/pull/125),源码及预览修复至 `6026b2d`,共享引用提交 `df90755`,基于 `d4c6437`;连接本地开关、三信号、Peer/Job carrier/Link、可选诊断链接与认证 relay | +| Grafana Cloud | Sir 已创建 stack 并配置写入与 Viewer 凭据;API 当前标记为 `free-trial`,不能据此宣称稳定 Free 套餐边界已验。三类合成数据已由 Viewer API 独立读回;Trace/Span/Link、AI unknown/zero、日志和指标均与源输入一致 | 持久行为见 [core-py 部署文档](../../docs/40-deployment/observability.md)和已发布[共享契约](../../docs/_shared/20-product-tdd/observability-contract.md)。任务内旧设计与实验保留其历史语境,不再作为运行事实的第二权威。 @@ -27,10 +27,16 @@ ## 剩余交付门禁 -本批 draft 不等于完整 G2/G3 或生产准入。正式客户端类型需在 core release artifact 准入后,使用既有 sync 流程重新生成并比较;不修改准入规则来容纳候选数据库。Hub 合并后依赖 Spoke 必须更新至合并后的共享 SHA。 +本批实现交付不等于完整 G2/G3 或持续生产遥测启用准入。正式客户端类型需在 core release artifact 准入后,使用既有 sync 流程重新生成并比较;不修改准入规则来容纳候选数据库。Hub 合并后两个 Spoke 的共享 SHA 已独立更新。 云端合成数据读回及 API 样本导出已通过,免费额度、保留边界与长期历史导出仍待验证;真实平台终止宽限与请求后冻结、真实模型 provider、活跃部署覆盖及 webext 独立生命周期仍是部署或后续覆盖门禁。AI 原文/长期快照没有被基础 tracing 自动授权。正常 SDK shutdown 的等待窗口不是端到端硬时限或零丢失承诺。 ## 资源与权限 -本轮 task-owned `inkcre-o11y-db-g1-b0a97f7c` 的两个容器和隧道已停止;`o11y_impl`、`o11y_migration_roundtrip`、历史 `o11y_lab` 与其余实验卷保留。其它三历史实验项目也保持停止。未操作 SVC 开发数据库;未升级套餐、合并或生产部署。真实凭据仅留本地忽略配置,不进入任务证据、提交或浏览器。父任务保持活跃。 +本轮 task-owned `inkcre-o11y-db-g1-b0a97f7c` 的两个容器和隧道已停止;`o11y_impl`、`o11y_migration_roundtrip`、历史 `o11y_lab` 与其余实验卷保留。其它三历史实验项目也保持停止。未操作 SVC 开发数据库;未升级套餐。Hub 已合并,应用交付推进中。真实凭据仅留本地忽略配置,不进入任务证据、提交或浏览器。父任务保持活跃。 + +## 当前预览验收 + +PR #124 独立 Heroku/Neon 预览曾临时开启 PG 日志与 OTLP,私密值仅通过 Heroku API 写入服务端,原配置保存于忽略的恢复文件。真实 Chromium 初次读取共享配置被 1.5 秒上限取消,业务 Job 与 PG 错误日志正常;独立 HTTP 读同一合法配置耗时 2.376 秒。客户端 `b64e5eb` 将配置读取上限设为 10 秒,直发/relay exporter 分别为 10/35 秒,处理器多留 5 秒;调用方 flush/shutdown 等待仍 1.5 秒。修复后真实 Job 三信号/Link 经 Cloud 独立读回通过,Pages 预览显示业务终态与 PG 日志。配置 10 秒超时、上游拒连、再次关闭后的业务/PG 保留已验;`6026b2d` 对齐指标标签及秒分桶,Cloud 严格读回通过。真实付费模型未调用。 + +预览运行配置与共享配置已恢复,新增私密出口变量已移除,见 `experiments/evidence/preview-restored.json`。预览原有显式 logging backend 为 none;临时 PG 日志只用于本次开关/故障验证,不修改生产配置。 diff --git a/tasks/observability-foundation/verification.md b/tasks/observability-foundation/verification.md index dbf4aa2..700c9fe 100644 --- a/tasks/observability-foundation/verification.md +++ b/tasks/observability-foundation/verification.md @@ -11,7 +11,7 @@ | V2 | 提交与另一 Peer 的 Job 执行可关联,重启不依赖内存上下文 | 提交后重启发起 Peer;另一执行 Peer 领取;核对持久提交上下文、执行 Span Link 和 Job 终态。覆盖 Python/TS 生产者与执行者、Cron、旧 Job 及开→关/关→开/关→关;关闭端创建 NULL,领取/关闭保留已有 carrier | 正式 migration、实际 Python/TS Job和mixed开关已验,关闭端保留carrier。独立迁移升级/降级/再升级及容量约束已验;生产协调升级未执行 | | V3 | 新增观测开关不改变现有 PG 日志能力 | 开关关闭/开启/后端中断/再次关闭时,从真实 Job 页核对当前与历史 PG 日志、job.、排序分页;新诊断链接为可选,旧 writer 不自动停写,原日志 trace_id 不替换为 OTel ID | Python真实Job/PG日志在开启与关闭路径保持;客户端Job页入口保持并有UI烟测。生产历史数据分页仍待部署验收 | | V4 | AI 耗时、结束原因、usage 与步骤关系可信 | 对受控非流式、流式末块、usage 缺失和并行工具案例比较 provider 原始回包与导出数据;再用授权 preview 验证一个真实 provider。未知不记零,估算成本不冒充账单 | 实际 OpenAI SDK+合成provider经生产adapter/Agent已验:31span/13HTTP请求,usage-only末块、重复累计、unknown/zero、并发错误取消。真实外部provider尚未调用 | -| V5 | 新增 OTLP 基础模式没有意外内容副本,采集权限不等于读取/管理权限 | 合成敏感 canary 经过请求、异常、工具结果与 SQL 参数路径后,检查实际导出数据;验证采集入口及读者边界。PG 按原配置保留,不能以此宣称整个部署无原文。新增内容模式另验保留、截断、删除与 blob 访问 | 实际OTLP出口canary与受控字段已验;relayJWT、私密头、256KiB/并发4/3秒已验;SDK内部指标View/exemplar去敏已验。旧PG原文行为保持 | +| V5 | 新增 OTLP 基础模式没有意外内容副本,采集权限不等于读取/管理权限 | 合成敏感 canary 经过请求、异常、工具结果与 SQL 参数路径后,检查实际导出数据;验证采集入口及读者边界。PG 按原配置保留,不能以此宣称整个部署无原文。新增内容模式另验保留、截断、删除与 blob 访问 | 实际OTLP出口canary与受控字段已验;relayJWT、私密头、256KiB/并发4/按信号有界超时已验;SDK内部指标View/exemplar去敏已验。旧PG原文行为保持 | | V6 | 遥测不可用不改变业务成功、失败、取消和资源关闭语义 | 同一受控负载对比正常与断开 SDK→SaaS 和实际存在的中间转发;检查数据库业务终态、响应、队列/内存上限、丢弃计数和恢复;不可用时不能阻止应用 ready | 正常/慢接收/拒连、真实Job成功失败取消、metrics-only已验;8项SDK故障注入通过;1024条积压产生原生 queue_full 计数;已记录SDK并发误差,不能用作精确损失账本,最后metrics回收export失败 | | V7 | 可更换后端并保留必要的诊断能力 | 仅改标准 endpoint/认证配置,将同一采集样本改投第二个兼容后端,业务采集/Job schema 不改;按 Job/Trace/AI 字段查询,核对 links;另做历史数据导出读回与配置迁移清单。只收到 OTLP 不能算通过 | 历史第二后端实验保留,新增实际浏览器/core三信号标准PBF已验;目标Cloud必要 Trace/Link/AI 字段、日志/metric查询及样本导出已通过;长期历史导出仍未验 | | V8 | 托管接入符合 scale-to-0 与当前零新增观测费,可维护和迁移 | 验 SDK 直发与短生命周期 flush/冻结、无 scrape 唤醒、应用附加计费时间;确认实际 Free、无收费依赖,量化摄取/查询/保留额度与限额失效,验删除/导出/配置重建和供应商恢复范围。自建时另验冷备与资源 | 实际stack已创建;Viewer查询API三类均200,写入认证已修正,实际三信号存储读回通过。默认无常驻Collector/scrape;真实免费套餐额度及平台freeze/终止宽限未验 | @@ -51,3 +51,7 @@ 本次 SaaS 修订的语法、链接、证据、Hub patch 与资源检查记录在 `experiments/evidence/sentinel-saas-20261003/packet-check.json`;原报告保持历史时间边界,不覆盖前轮证据。 历史 D10 阶段只做显式 opt-in、PG 保留和 vendor-agnostic 候选差异的任务包/Hub patch 静态检查;当时未运行新应用实验。当前 V0/V3 结论以本轮实现证据为准。检查见 `experiments/evidence/sentinel-saas-20261003/opt-in-packet-check.json`。 + +## 预览交付补验 + +真实 Chromium 业务模块→PR #124 PostgREST→Heroku Core Job 执行→Grafana 三信号已闭环,提交/执行 IDs 与 Link 经 Viewer API 独立核对,实际 Pages 预览已显示业务终态与 PG 日志。第一次 1.5 秒配置读取超时被保留为失败证据,客户端修复后通过。精确运行范围与 Free 试用状态见 [Cloud 验收](experiments/cloud-acceptance.md)。此项不冒充 Pages UI 发起 Job、真实模型、长期保留或生产平台完全验收。