From 9a642cdb716fa12b69d246374e9c4c6f63590da0 Mon Sep 17 00:00:00 2001 From: zhenyu Date: Fri, 29 Aug 2025 19:33:31 +0800 Subject: [PATCH] =?UTF-8?q?=E6=9B=B4=E6=96=B0=E6=9C=80=E6=96=B0=20main=20e?= =?UTF-8?q?n=20=E7=89=88=E6=9C=AC=E5=B8=AE=E5=8A=A9=E6=96=87=E6=A1=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/zh/01-about/06-users.md | 84 +-- docs/zh/02-ce-install/01-overview.md | 2 +- docs/zh/02-ce-install/02-all-in-one.md | 3 +- docs/zh/02-ce-install/08-serverless-pod.md | 1 + docs/zh/02-ce-install/09-ai-agent.md | 51 +- .../03-special-environment-deployment.md | 2 +- .../07-storage-engine-use-byconity.md | 184 +++---- .../01-l7-protocols/01-overview.md | 1 + docs/zh/05-features/01-l7-protocols/03-rpc.md | 32 +- .../05-features/01-l7-protocols/05-nosql.md | 31 +- docs/zh/05-features/01-l7-protocols/06-mq.md | 32 +- .../05-features/01-l7-protocols/07-network.md | 28 +- .../zh/05-features/01-l7-protocols/08-otel.md | 2 +- .../01-l7-protocols/09-skywalking.md | 52 +- .../07-metrics-and-operators.md | 10 +- .../01-auto-profiling.md | 2 + .../05-auto-tagging/04-custom-tags.md | 4 +- .../05-features/05-auto-tagging/08-k8s-crd.md | 22 +- .../06-guide/01-quick-start/01-5w-method.md | 8 + .../01-process/01-wasm-plugin.md | 1 + .../02-input/01-metrics/02-prometheus.md | 4 +- .../02-input/02-tracing/03-apm-trace-api.md | 42 +- .../02-input/04-log/02-vector.md | 10 +- .../03-output/01-query/04-mcp-server.md | 14 +- docs/zh/10-release-notes/08-ee-6.6-release.md | 8 +- translate/translated/01-about/06-users.md | 76 ++- .../translated/02-ce-install/01-overview.md | 94 ++-- .../translated/02-ce-install/02-all-in-one.md | 75 +-- .../translated/02-ce-install/03-single-k8s.md | 39 +- .../translated/02-ce-install/04-multi-k8s.md | 32 +- .../02-ce-install/05-legacy-host.md | 64 ++- .../02-ce-install/08-serverless-pod.md | 30 +- .../translated/02-ce-install/09-ai-agent.md | 178 +++++-- .../translated/02-ce-install/99-upgrade.md | 49 +- .../03-ee-install/01-saas/01-cloud.md | 147 +++--- .../01-agent-advanced-config.md | 45 +- .../03-special-environment-deployment.md | 171 +++--- .../04-reduce-storage-overhead.md | 257 ++++----- .../06-production-deployment.md | 94 ++-- .../07-storage-engine-use-byconity.md | 374 +++++++++++++ .../07-trouble-shooting-flow.md | 274 ---------- .../01-l7-protocols/01-overview.md | 17 +- .../05-features/01-l7-protocols/03-rpc.md | 408 +++++++-------- .../05-features/01-l7-protocols/05-nosql.md | 172 +++--- .../05-features/01-l7-protocols/06-mq.md | 490 ++++++++++-------- .../05-features/01-l7-protocols/07-network.md | 71 ++- .../05-features/01-l7-protocols/08-otel.md | 100 ++-- .../01-l7-protocols/09-skywalking.md | 38 ++ .../07-metrics-and-operators.md | 351 ++++++++----- .../01-auto-profiling.md | 123 +++-- .../05-auto-tagging/04-custom-tags.md | 54 +- .../06-additional-cloud-tags.md | 259 ++++----- .../05-features/05-auto-tagging/08-k8s-crd.md | 149 +++--- .../01-ee-tenant/01-query/01-overview.md | 23 - .../01-query/02-service-search.md | 115 ---- .../01-ee-tenant/01-query/04-log-search.md | 79 --- .../01-ee-tenant/01-query/05-metric-search.md | 23 - .../01-ee-tenant/01-query/06-history.md | 51 -- .../01-query/07-left-quick-filter.md | 32 -- .../01-ee-tenant/02-dashboard/02-list.md | 26 - .../01-ee-tenant/02-dashboard/03-use.md | 72 --- .../01-ee-tenant/02-dashboard/04-add-panel.md | 24 - .../02-dashboard/05-variable-template.md | 110 ---- .../02-dashboard/06-right-slide-box.md | 40 -- .../02-dashboard/99-panel/01-overview.md | 22 - .../02-dashboard/99-panel/02-topology.md | 153 ------ .../02-dashboard/99-panel/03-flame.md | 139 ----- .../02-dashboard/99-panel/04-line.md | 93 ---- .../02-dashboard/99-panel/05-bar.md | 31 -- .../02-dashboard/99-panel/06-pie.md | 17 - .../02-dashboard/99-panel/07-histogram.md | 17 - .../02-dashboard/99-panel/08-table.md | 80 --- .../02-dashboard/99-panel/09-stat.md | 38 -- .../02-dashboard/99-panel/10-text.md | 17 - .../03-universal-map/01-overview.md | 16 - .../03-universal-map/02-business-def.md | 112 ---- .../03-universal-map/03-service-list.md | 23 - .../03-universal-map/04-service-map.md | 97 ---- .../01-ee-tenant/04-tracing/01-overview.md | 20 - .../04-tracing/02-service-list.md | 32 -- .../04-tracing/03-service-statistics.md | 30 -- .../04-tracing/04-path-topology.md | 26 - .../01-ee-tenant/04-tracing/05-call-log.md | 28 - .../04-tracing/06-call-chain-tracing.md | 27 - .../04-tracing/07-file-reading-and-writing.md | 14 - .../04-tracing/08-right-sliding-box.md | 210 -------- .../05-profiling/01-continue-profile.md | 40 -- .../01-ee-tenant/06-network/01-overview.md | 21 - .../06-network/02-service-statistics.md | 16 - .../06-network/03-network-path.md | 16 - .../01-ee-tenant/06-network/04-network-map.md | 16 - .../01-ee-tenant/06-network/05-flow-log.md | 16 - .../06-network/06-NAT-traversal.md | 19 - .../06-network/07-resource-inventory.md | 16 - .../06-network/08-pacp-strategy.md | 38 -- .../06-network/09-pcap-download.md | 22 - .../01-ee-tenant/07-metrics/01-overview.md | 18 - .../01-ee-tenant/07-metrics/02-host.md | 16 - .../01-ee-tenant/07-metrics/03-container.md | 18 - .../07-metrics/05-metric-summary.md | 15 - .../07-metrics/06-metrics-template.md | 34 -- .../06-guide/01-ee-tenant/08-log/01-log.md | 26 - .../01-ee-tenant/09-alert/01-overview.md | 16 - .../01-ee-tenant/09-alert/02-alert-policy.md | 45 -- .../01-ee-tenant/09-alert/03-push-endpoint.md | 100 ---- .../01-ee-tenant/09-alert/04-alert-event.md | 17 - .../01-ee-tenant/10-report/01-report.md | 47 -- .../01-ee-tenant/11-resources/02-summary.md | 15 - .../11-resources/03-resource-changes.md | 16 - .../11-resources/04-resource-pool.md | 54 -- .../11-resources/05-computing-resources.md | 30 -- .../11-resources/06-network-resources.md | 49 -- .../11-resources/07-network-services.md | 45 -- .../11-resources/08-storage-services.md | 18 - .../11-resources/09-container-resources.md | 56 -- .../11-resources/11-other-resources.md | 33 -- .../01-ee-tenant/12-system/01-overview.md | 15 - .../01-ee-tenant/12-system/02-agent.md | 89 ---- .../01-ee-tenant/12-system/03-data-node.md | 30 -- .../12-system/05-operation-log.md | 13 - .../13-configuration/01-settings.md | 43 -- .../06-guide/01-quick-start/01-5w-method.md | 312 +++++++++++ .../02-ee-tenant/01-query/01-overview.md | 23 + .../01-query/02-service-search.md | 117 +++++ .../01-query/03-path-search.md | 64 +-- .../02-ee-tenant/01-query/04-log-search.md | 79 +++ .../02-ee-tenant/01-query/05-metric-search.md | 23 + .../02-ee-tenant/01-query/06-history.md | 51 ++ .../01-query/07-left-quick-filter.md | 32 ++ .../02-dashboard/01-overview.md | 6 +- .../02-ee-tenant/02-dashboard/02-list.md | 26 + .../02-ee-tenant/02-dashboard/03-use.md | 72 +++ .../02-ee-tenant/02-dashboard/04-add-panel.md | 24 + .../02-dashboard/05-variable-template.md | 114 ++++ .../02-dashboard/06-right-slide-box.md | 40 ++ .../02-dashboard/99-panel/01-overview.md | 22 + .../02-dashboard/99-panel/02-topology.md | 153 ++++++ .../02-dashboard/99-panel/03-flame.md | 139 +++++ .../02-dashboard/99-panel/04-line.md | 93 ++++ .../02-dashboard/99-panel/05-bar.md | 31 ++ .../02-dashboard/99-panel/06-pie.md | 17 + .../02-dashboard/99-panel/07-histogram.md | 17 + .../02-dashboard/99-panel/08-table.md | 80 +++ .../02-dashboard/99-panel/09-stat.md | 38 ++ .../02-dashboard/99-panel/10-text.md | 17 + .../03-universal-map/01-overview.md | 16 + .../03-universal-map/02-business-def.md | 112 ++++ .../03-universal-map/03-service-list.md | 23 + .../03-universal-map/04-service-map.md | 97 ++++ .../02-ee-tenant/04-tracing/01-overview.md | 20 + .../04-tracing/02-service-list.md | 32 ++ .../04-tracing/03-service-statistics.md | 30 ++ .../04-tracing/04-path-topology.md | 26 + .../02-ee-tenant/04-tracing/05-call-log.md | 28 + .../04-tracing/06-call-chain-tracing.md | 27 + .../04-tracing/07-file-reading-and-writing.md | 14 + .../04-tracing/08-right-sliding-box.md | 210 ++++++++ .../05-profiling/01-continue-profile.md | 40 ++ .../02-ee-tenant/06-network/01-overview.md | 21 + .../06-network/02-service-statistics.md | 16 + .../06-network/03-network-path.md | 16 + .../02-ee-tenant/06-network/04-network-map.md | 16 + .../02-ee-tenant/06-network/05-flow-log.md | 16 + .../06-network/06-NAT-traversal.md | 19 + .../06-network/07-resource-inventory.md | 16 + .../06-network/08-pacp-strategy.md | 38 ++ .../06-network/09-pcap-download.md | 22 + .../02-ee-tenant/07-metrics/01-overview.md | 18 + .../02-ee-tenant/07-metrics/02-host.md | 16 + .../02-ee-tenant/07-metrics/03-container.md | 18 + .../07-metrics/04-metrics-viewing.md | 10 +- .../07-metrics/05-metric-summary.md | 15 + .../07-metrics/06-metrics-template.md | 34 ++ .../06-guide/02-ee-tenant/08-log/01-log.md | 26 + .../02-ee-tenant/09-alert/01-overview.md | 16 + .../02-ee-tenant/09-alert/02-alert-policy.md | 45 ++ .../02-ee-tenant/09-alert/03-push-endpoint.md | 100 ++++ .../02-ee-tenant/09-alert/04-alert-event.md | 17 + .../02-ee-tenant/10-report/01-report.md | 47 ++ .../11-resources/01-overview.md | 2 +- .../02-ee-tenant/11-resources/02-summary.md | 15 + .../11-resources/03-resource-changes.md | 16 + .../11-resources/04-resource-pool.md | 54 ++ .../11-resources/05-computing-resources.md | 30 ++ .../11-resources/06-network-resources.md | 49 ++ .../11-resources/07-network-services.md | 46 ++ .../11-resources/08-storage-services.md | 18 + .../11-resources/09-container-resources.md | 56 ++ .../11-resources/10-process-resources.md | 4 +- .../11-resources/11-other-resources.md | 33 ++ .../02-ee-tenant/12-system/01-overview.md | 15 + .../02-ee-tenant/12-system/02-agent.md | 89 ++++ .../02-ee-tenant/12-system/03-data-node.md | 30 ++ .../12-system/04-account-management.md | 2 +- .../12-system/05-operation-log.md | 13 + .../13-configuration/01-settings.md | 43 ++ .../translated/07-configuration/README.md | 12 +- .../01-process/01-wasm-plugin.md | 149 +++--- .../01-metrics/01-metrics-auto-tagging.md | 4 +- .../02-input/01-metrics/02-prometheus.md | 80 ++- .../02-input/01-metrics/03-telegraf.md | 42 +- .../02-input/01-metrics/04-grafana-agent.md | 16 +- .../01-full-stack-distributed-tracing.md | 28 +- .../02-tracing/02-traccing-auto-tagging.md | 4 +- .../02-input/02-tracing/03-apm-trace-api.md | 36 +- .../02-input/02-tracing/04-opentelemetry.md | 77 +-- .../02-input/02-tracing/05-skywalking.md | 93 +++- .../03-profile/01-profile-auto-tagging.md | 4 +- .../02-input/03-profile/02-profile.md | 60 +-- .../02-input/04-log/01-log-auto-tagging.md | 4 +- .../02-input/04-log/02-vector.md | 105 +++- .../03-output/01-query/01-sql.md | 83 +-- .../03-output/01-query/02-promql.md | 62 +-- .../03-output/01-query/03-trace-completion.md | 142 ++--- .../03-output/01-query/04-mcp-server.md | 62 +++ .../02-export/01-opentelemetry-exporter.md | 422 +++++++-------- .../02-export/02-prom-remote-write.md | 44 +- .../03-output/02-export/03-kafka-exporter.md | 30 +- .../03-output/02-export/04-exporter-config.md | 64 +-- translate/translated/09-diagnose/01-FAQ.md | 50 +- .../translated/09-diagnose/02-grafana.md | 20 +- .../09-diagnose/03-deepflow-agent.md | 10 +- .../09-diagnose/04-deepflow-server.md | 114 ++-- .../10-release-notes/01-versioning.md | 14 +- .../10-release-notes/02-release-timeline.md | 44 +- .../10-release-notes/03-ce-6.6-release.md | 83 --- .../10-release-notes/03-ce-7.1-release.md | 24 + .../10-release-notes/04-ee-6.6-release.md | 8 - .../10-release-notes/04-ee-7.1-release.md | 68 +++ .../10-release-notes/05-ce-7.0-release.md | 120 +++++ .../10-release-notes/06-ee-7.0-release.md | 118 +++++ .../10-release-notes/07-ce-6.6-release.md | 254 +++++++++ .../10-release-notes/08-ee-6.6-release.md | 325 ++++++++++++ ...ce-6.5-release.md => 09-ce-6.5-release.md} | 0 ...ee-6.5-release.md => 10-ee-6.5-release.md} | 0 ...ce-6.4-release.md => 11-ce-6.4-release.md} | 0 ...ee-6.4-release.md => 12-ee-6.4-release.md} | 0 ...ce-6.3-release.md => 13-ce-6.3-release.md} | 0 ...ee-6.3-release.md => 14-ee-6.3-release.md} | 0 ...ce-6.2-release.md => 15-ce-6.2-release.md} | 0 ...ee-6.2-release.md => 16-ee-6.2-release.md} | 0 ...ce-6.1-release.md => 17-ce-6.1-release.md} | 0 ...ee-6.1-release.md => 18-ee-6.1-release.md} | 0 translate/translated/README.md | 2 + 244 files changed, 8020 insertions(+), 5955 deletions(-) create mode 100644 translate/translated/04-best-practice/07-storage-engine-use-byconity.md delete mode 100644 translate/translated/04-best-practice/07-trouble-shooting-flow.md create mode 100644 translate/translated/05-features/01-l7-protocols/09-skywalking.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/01-query/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/01-query/02-service-search.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/01-query/04-log-search.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/01-query/05-metric-search.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/01-query/06-history.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/01-query/07-left-quick-filter.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/02-list.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/03-use.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/04-add-panel.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/05-variable-template.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/06-right-slide-box.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/02-topology.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/03-flame.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/04-line.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/05-bar.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/06-pie.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/07-histogram.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/08-table.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/09-stat.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/10-text.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/03-universal-map/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/03-universal-map/02-business-def.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/03-universal-map/03-service-list.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/03-universal-map/04-service-map.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/02-service-list.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/03-service-statistics.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/04-path-topology.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/05-call-log.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/06-call-chain-tracing.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/07-file-reading-and-writing.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/04-tracing/08-right-sliding-box.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/05-profiling/01-continue-profile.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/02-service-statistics.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/03-network-path.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/04-network-map.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/05-flow-log.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/06-NAT-traversal.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/07-resource-inventory.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/08-pacp-strategy.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/06-network/09-pcap-download.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/07-metrics/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/07-metrics/02-host.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/07-metrics/03-container.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/07-metrics/05-metric-summary.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/07-metrics/06-metrics-template.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/08-log/01-log.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/09-alert/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/09-alert/02-alert-policy.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/09-alert/03-push-endpoint.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/09-alert/04-alert-event.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/10-report/01-report.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/02-summary.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/03-resource-changes.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/04-resource-pool.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/05-computing-resources.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/06-network-resources.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/07-network-services.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/08-storage-services.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/09-container-resources.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/11-resources/11-other-resources.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/12-system/01-overview.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/12-system/02-agent.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/12-system/03-data-node.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/12-system/05-operation-log.md delete mode 100644 translate/translated/06-guide/01-ee-tenant/13-configuration/01-settings.md create mode 100644 translate/translated/06-guide/01-quick-start/01-5w-method.md create mode 100644 translate/translated/06-guide/02-ee-tenant/01-query/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/01-query/02-service-search.md rename translate/translated/06-guide/{01-ee-tenant => 02-ee-tenant}/01-query/03-path-search.md (66%) create mode 100644 translate/translated/06-guide/02-ee-tenant/01-query/04-log-search.md create mode 100644 translate/translated/06-guide/02-ee-tenant/01-query/05-metric-search.md create mode 100644 translate/translated/06-guide/02-ee-tenant/01-query/06-history.md create mode 100644 translate/translated/06-guide/02-ee-tenant/01-query/07-left-quick-filter.md rename translate/translated/06-guide/{01-ee-tenant => 02-ee-tenant}/02-dashboard/01-overview.md (54%) create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/02-list.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/03-use.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/04-add-panel.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/05-variable-template.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/06-right-slide-box.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/02-topology.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/03-flame.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/04-line.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/05-bar.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/06-pie.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/07-histogram.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/08-table.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/09-stat.md create mode 100644 translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/10-text.md create mode 100644 translate/translated/06-guide/02-ee-tenant/03-universal-map/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/03-universal-map/02-business-def.md create mode 100644 translate/translated/06-guide/02-ee-tenant/03-universal-map/03-service-list.md create mode 100644 translate/translated/06-guide/02-ee-tenant/03-universal-map/04-service-map.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/02-service-list.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/03-service-statistics.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/04-path-topology.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/05-call-log.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/06-call-chain-tracing.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/07-file-reading-and-writing.md create mode 100644 translate/translated/06-guide/02-ee-tenant/04-tracing/08-right-sliding-box.md create mode 100644 translate/translated/06-guide/02-ee-tenant/05-profiling/01-continue-profile.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/02-service-statistics.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/03-network-path.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/04-network-map.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/05-flow-log.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/06-NAT-traversal.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/07-resource-inventory.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/08-pacp-strategy.md create mode 100644 translate/translated/06-guide/02-ee-tenant/06-network/09-pcap-download.md create mode 100644 translate/translated/06-guide/02-ee-tenant/07-metrics/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/07-metrics/02-host.md create mode 100644 translate/translated/06-guide/02-ee-tenant/07-metrics/03-container.md rename translate/translated/06-guide/{01-ee-tenant => 02-ee-tenant}/07-metrics/04-metrics-viewing.md (51%) create mode 100644 translate/translated/06-guide/02-ee-tenant/07-metrics/05-metric-summary.md create mode 100644 translate/translated/06-guide/02-ee-tenant/07-metrics/06-metrics-template.md create mode 100644 translate/translated/06-guide/02-ee-tenant/08-log/01-log.md create mode 100644 translate/translated/06-guide/02-ee-tenant/09-alert/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/09-alert/02-alert-policy.md create mode 100644 translate/translated/06-guide/02-ee-tenant/09-alert/03-push-endpoint.md create mode 100644 translate/translated/06-guide/02-ee-tenant/09-alert/04-alert-event.md create mode 100644 translate/translated/06-guide/02-ee-tenant/10-report/01-report.md rename translate/translated/06-guide/{01-ee-tenant => 02-ee-tenant}/11-resources/01-overview.md (68%) create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/02-summary.md create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/03-resource-changes.md create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/04-resource-pool.md create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/05-computing-resources.md create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/06-network-resources.md create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/07-network-services.md create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/08-storage-services.md create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/09-container-resources.md rename translate/translated/06-guide/{01-ee-tenant => 02-ee-tenant}/11-resources/10-process-resources.md (58%) create mode 100644 translate/translated/06-guide/02-ee-tenant/11-resources/11-other-resources.md create mode 100644 translate/translated/06-guide/02-ee-tenant/12-system/01-overview.md create mode 100644 translate/translated/06-guide/02-ee-tenant/12-system/02-agent.md create mode 100644 translate/translated/06-guide/02-ee-tenant/12-system/03-data-node.md rename translate/translated/06-guide/{01-ee-tenant => 02-ee-tenant}/12-system/04-account-management.md (71%) create mode 100644 translate/translated/06-guide/02-ee-tenant/12-system/05-operation-log.md create mode 100644 translate/translated/06-guide/02-ee-tenant/13-configuration/01-settings.md create mode 100644 translate/translated/08-integration/03-output/01-query/04-mcp-server.md delete mode 100644 translate/translated/10-release-notes/03-ce-6.6-release.md create mode 100644 translate/translated/10-release-notes/03-ce-7.1-release.md delete mode 100644 translate/translated/10-release-notes/04-ee-6.6-release.md create mode 100644 translate/translated/10-release-notes/04-ee-7.1-release.md create mode 100644 translate/translated/10-release-notes/05-ce-7.0-release.md create mode 100644 translate/translated/10-release-notes/06-ee-7.0-release.md create mode 100644 translate/translated/10-release-notes/07-ce-6.6-release.md create mode 100644 translate/translated/10-release-notes/08-ee-6.6-release.md rename translate/translated/10-release-notes/{05-ce-6.5-release.md => 09-ce-6.5-release.md} (100%) rename translate/translated/10-release-notes/{06-ee-6.5-release.md => 10-ee-6.5-release.md} (100%) rename translate/translated/10-release-notes/{07-ce-6.4-release.md => 11-ce-6.4-release.md} (100%) rename translate/translated/10-release-notes/{08-ee-6.4-release.md => 12-ee-6.4-release.md} (100%) rename translate/translated/10-release-notes/{09-ce-6.3-release.md => 13-ce-6.3-release.md} (100%) rename translate/translated/10-release-notes/{10-ee-6.3-release.md => 14-ee-6.3-release.md} (100%) rename translate/translated/10-release-notes/{11-ce-6.2-release.md => 15-ce-6.2-release.md} (100%) rename translate/translated/10-release-notes/{12-ee-6.2-release.md => 16-ee-6.2-release.md} (100%) rename translate/translated/10-release-notes/{13-ce-6.1-release.md => 17-ce-6.1-release.md} (100%) rename translate/translated/10-release-notes/{14-ee-6.1-release.md => 18-ee-6.1-release.md} (100%) diff --git a/docs/zh/01-about/06-users.md b/docs/zh/01-about/06-users.md index 6578aa80..f91eb985 100644 --- a/docs/zh/01-about/06-users.md +++ b/docs/zh/01-about/06-users.md @@ -5,54 +5,54 @@ permalink: /about/users # 2025 -| 行业 | 用户 | 来源 | 标题 | 链接 | -| -------- | ---------- | ------ | ------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| 云服务 | 腾讯云 | Meetup | DeepFlow 在腾讯 TKE 内部平台的可观测性实践 | [文章](https://mp.weixin.qq.com/s/tsVObqnUBOQ-fE6uK6_oxA),[PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/85bfdb75cf8v77d8b618bf2h90a769b4_20241217152025.pdf),[视频](https://www.bilibili.com/video/BV1y2kEYkEf1) | -| 新制造 | 某智能车企 | 文章 | 慢调用排查实录:高效定界服务网格 Sidecar 性能瓶颈 | [文章](https://mp.weixin.qq.com/s/0QtKqQDuV1KjYYCPAP4sBQ) | -| 互联网 | 金山办公 | Meetup | 金山办公基于 DeepFlow docker 模式的可观测性实践 | [文章](https://mp.weixin.qq.com/s/Pd1-lO9pjAKhofzmpuEBgw),[PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/jpg/6815b7d64b37417156ca9f8bf1403705_20250702104904.pdf),[视频](https://www.bilibili.com/video/BV1AogSz9E3o) | +| 行业 | 用户 | 来源 | 标题 | 链接 | +| ------ | ---------- | ------ | ------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 云服务 | 腾讯云 | Meetup | DeepFlow 在腾讯 TKE 内部平台的可观测性实践 | [文章](https://mp.weixin.qq.com/s/tsVObqnUBOQ-fE6uK6_oxA),[PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/85bfdb75cf8v77d8b618bf2h90a769b4_20241217152025.pdf),[视频](https://www.bilibili.com/video/BV1y2kEYkEf1) | +| 新制造 | 某智能车企 | 文章 | 慢调用排查实录:高效定界服务网格 Sidecar 性能瓶颈 | [文章](https://mp.weixin.qq.com/s/0QtKqQDuV1KjYYCPAP4sBQ) | +| 互联网 | 金山办公 | Meetup | 金山办公基于 DeepFlow docker 模式的可观测性实践 | [文章](https://mp.weixin.qq.com/s/Pd1-lO9pjAKhofzmpuEBgw),[PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/jpg/6815b7d64b37417156ca9f8bf1403705_20250702104904.pdf),[视频](https://www.bilibili.com/video/BV1AogSz9E3o) | # 2024 -| 行业 | 用户 | 来源 | 标题 | 链接 | -| -------- | ---------- | ------ | ------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| 金融 | 某头部券商 | 文章 | 可观测性实战:从拨云见日到抽丝剥茧快速定位业务响应时延高问题 | [文章](https://mp.weixin.qq.com/s/4ZRfRlgHw2DWaSuFaHiJDg) | -| 银行 | 某股份银行 | 文章 | eBPF 零侵扰分布式追踪 3 分钟锁定 Java 程序 I/O 线程阻塞 | [文章](https://mp.weixin.qq.com/s/8998CkTGrvPwoad5wkqitA) | -| 银行 | 某股份银行 | 文章 | 故障诊断 3 分钟锁定分布式核心数据库,加速金融科技信创开发、测试、迁移 | [文章](https://mp.weixin.qq.com/s/VfoPeKp-iMeQJc2VMEAZtA) | -| 银行 | 某国有银行 | 文章 | 3 分钟诊断 Tomcat TCP 超时参数配置错误引发的概率性交易失败 | [文章](https://mp.weixin.qq.com/s/lao6SRU6xwo0ImEAqlNbfQ) | -| 银行 | 某股份银行 | 文章 | DeepFlow 大模型智能体 3 分钟定位 Java 程序 Hang 故障 | [文章](https://mp.weixin.qq.com/s/1H3mqKRL0GBrE3qstfqKlQ) | -| 电信 | 中国移动 | 文章 | 深度解析 DeepFlow 如何采集大模型服务的业务指标 | [文章](https://mp.weixin.qq.com/s/GjIKMIaDxbxNo75uhvTgAg) | -| 游戏 | 腾讯互娱 | Meetup | 蓝鲸观测平台:统一观测数据关联模型探索 | [文章](https://mp.weixin.qq.com/s/-osVmY1V6yAycVx6d3CAMA), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/721e3ac62b51e6234eb10f03e7d41629_20240914102155.pdf), [视频](https://www.bilibili.com/video/BV1ZJ46eiE3S) | -| 互联网 | 金山办公 | Meetup | 金山办公基于 DeepFlow 的零侵扰可观测性实践 | [文章](https://mp.weixin.qq.com/s/7M5BCzDDQ3NQmuieIeCqXw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/3838d594c942dc4765a223573206e5b5_20240913152317.pdf), [视频](https://www.bilibili.com/video/BV1JV46e6ErU) | -| 金融 | 富途证券 | 文章 | 从部署到优化:富途证券的 DeepFlow 探索之旅 | [文章](https://mp.weixin.qq.com/s/xFBiyTRrADUnCOMPwrTOQw) | -| 金融 | FinPoints | 文章 | FinPoints x DeepFlow:如何实现 SRE 99.9% 服务级别目标 (SLO) | [文章](https://mp.weixin.qq.com/s/WoGDcmT1ua3N3DXa11Bk4g) | -| 互联网 | 企迈科技 | 文章 | 企迈科技 x DeepFlow:爆发式增长业务背后的可观测性平台实践 | [文章](https://mp.weixin.qq.com/s/P2tMeAYCMns05zG8nfj6dg) | -| 消费电子 | 某手机厂商 | 文章 | 使用 DeepFlow 消除 APISIX 故障诊断中的“南辕北辙” | [文章](https://mp.weixin.qq.com/s/a-x_ce6VO-L1SaXs8PKoAg) | -| 云服务 | 腾讯云 | Meetup | 腾讯云某业务基于 DeepFlow 的可观测性实践 | [文章](https://mp.weixin.qq.com/s/57e3dAvN9gYcwWGjt-BMbw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/52a0ea94c84600ddc34c53e10e048420_20240802114858.pdf), [视频](https://www.bilibili.com/video/BV1q4421Z7ni) | -| 游戏 | 腾讯互娱 | 文章 | 腾讯游戏基于 DeepFlow 的零侵扰可观测性进阶实战 | [文章](https://mp.weixin.qq.com/s/6v5jPLSMD1SZJITIKvHpWA) | -| 电信 | 中国移动 | 文章 | 开箱即用的 eBPF 可观测性:中国移动磐基 PaaS 平台案例 | [文章](https://mp.weixin.qq.com/s/Byb_PJ7hlUAeTotAamgqRA) | -| 银行 | 某国有银行 | 文章 | DeepFlow 零侵扰实现分布式数据库 TDSQL 的全链路可观测性 | [文章](https://mp.weixin.qq.com/s/IJntZDqBpLOWP2-JGY6Hmw) | -| 电信 | 中国移动 | Meetup | DeepFlow 元数据数据库 PostgreSQL 改造实践 | [文章](https://mp.weixin.qq.com/s/1_8939kNHZjqrABB9nlzBg), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/713b09f77232c733ff17d2e81955d9f6_20240802124302.pdf), [视频](https://www.bilibili.com/video/BV1tZ421N7zQ) | -| 银行 | 民生银行 | Meetup | 民生银行云原生业务的 eBPF 可观测性建设实践 | [文章](https://mp.weixin.qq.com/s/9XctB-EPqOPSbK1YL2JzlQ), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ebae4e2d4d0ea71c28228c5e0dbb8f23_20231225162831.pdf), [视频](https://www.bilibili.com/video/BV1ag4y1C7DD) | -| 游戏 | 腾讯互娱 | Meetup | 消灭盲点!腾讯游戏真·全栈观测实践 | [文章](https://mp.weixin.qq.com/s/vzRebv7TMrrRi8TUV9qj5A), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/580f8117457f0e2bbc2f3818f7d42300_20231225162841.pdf), [视频](https://www.bilibili.com/video/BV1ku4y1K7PF) | +| 行业 | 用户 | 来源 | 标题 | 链接 | +| -------- | ---------- | ------ | --------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 金融 | 某头部券商 | 文章 | 可观测性实战:从拨云见日到抽丝剥茧快速定位业务响应时延高问题 | [文章](https://mp.weixin.qq.com/s/4ZRfRlgHw2DWaSuFaHiJDg) | +| 银行 | 某股份银行 | 文章 | eBPF 零侵扰分布式追踪 3 分钟锁定 Java 程序 I/O 线程阻塞 | [文章](https://mp.weixin.qq.com/s/8998CkTGrvPwoad5wkqitA) | +| 银行 | 某股份银行 | 文章 | 故障诊断 3 分钟锁定分布式核心数据库,加速金融科技信创开发、测试、迁移 | [文章](https://mp.weixin.qq.com/s/VfoPeKp-iMeQJc2VMEAZtA) | +| 银行 | 某国有银行 | 文章 | 3 分钟诊断 Tomcat TCP 超时参数配置错误引发的概率性交易失败 | [文章](https://mp.weixin.qq.com/s/lao6SRU6xwo0ImEAqlNbfQ) | +| 银行 | 某股份银行 | 文章 | DeepFlow 大模型智能体 3 分钟定位 Java 程序 Hang 故障 | [文章](https://mp.weixin.qq.com/s/1H3mqKRL0GBrE3qstfqKlQ) | +| 电信 | 中国移动 | 文章 | 深度解析 DeepFlow 如何采集大模型服务的业务指标 | [文章](https://mp.weixin.qq.com/s/GjIKMIaDxbxNo75uhvTgAg) | +| 游戏 | 腾讯互娱 | Meetup | 蓝鲸观测平台:统一观测数据关联模型探索 | [文章](https://mp.weixin.qq.com/s/-osVmY1V6yAycVx6d3CAMA), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/721e3ac62b51e6234eb10f03e7d41629_20240914102155.pdf), [视频](https://www.bilibili.com/video/BV1ZJ46eiE3S) | +| 互联网 | 金山办公 | Meetup | 金山办公基于 DeepFlow 的零侵扰可观测性实践 | [文章](https://mp.weixin.qq.com/s/7M5BCzDDQ3NQmuieIeCqXw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/3838d594c942dc4765a223573206e5b5_20240913152317.pdf), [视频](https://www.bilibili.com/video/BV1JV46e6ErU) | +| 金融 | 富途证券 | 文章 | 从部署到优化:富途证券的 DeepFlow 探索之旅 | [文章](https://mp.weixin.qq.com/s/xFBiyTRrADUnCOMPwrTOQw) | +| 金融 | FinPoints | 文章 | FinPoints x DeepFlow:如何实现 SRE 99.9% 服务级别目标 (SLO) | [文章](https://mp.weixin.qq.com/s/WoGDcmT1ua3N3DXa11Bk4g) | +| 互联网 | 企迈科技 | 文章 | 企迈科技 x DeepFlow:爆发式增长业务背后的可观测性平台实践 | [文章](https://mp.weixin.qq.com/s/P2tMeAYCMns05zG8nfj6dg) | +| 消费电子 | 某手机厂商 | 文章 | 使用 DeepFlow 消除 APISIX 故障诊断中的“南辕北辙” | [文章](https://mp.weixin.qq.com/s/a-x_ce6VO-L1SaXs8PKoAg) | +| 云服务 | 腾讯云 | Meetup | 腾讯云某业务基于 DeepFlow 的可观测性实践 | [文章](https://mp.weixin.qq.com/s/57e3dAvN9gYcwWGjt-BMbw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/52a0ea94c84600ddc34c53e10e048420_20240802114858.pdf), [视频](https://www.bilibili.com/video/BV1q4421Z7ni) | +| 游戏 | 腾讯互娱 | 文章 | 腾讯游戏基于 DeepFlow 的零侵扰可观测性进阶实战 | [文章](https://mp.weixin.qq.com/s/6v5jPLSMD1SZJITIKvHpWA) | +| 电信 | 中国移动 | 文章 | 开箱即用的 eBPF 可观测性:中国移动磐基 PaaS 平台案例 | [文章](https://mp.weixin.qq.com/s/Byb_PJ7hlUAeTotAamgqRA) | +| 银行 | 某国有银行 | 文章 | DeepFlow 零侵扰实现分布式数据库 TDSQL 的全链路可观测性 | [文章](https://mp.weixin.qq.com/s/IJntZDqBpLOWP2-JGY6Hmw) | +| 电信 | 中国移动 | Meetup | DeepFlow 元数据数据库 PostgreSQL 改造实践 | [文章](https://mp.weixin.qq.com/s/1_8939kNHZjqrABB9nlzBg), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/713b09f77232c733ff17d2e81955d9f6_20240802124302.pdf), [视频](https://www.bilibili.com/video/BV1tZ421N7zQ) | +| 银行 | 民生银行 | Meetup | 民生银行云原生业务的 eBPF 可观测性建设实践 | [文章](https://mp.weixin.qq.com/s/9XctB-EPqOPSbK1YL2JzlQ), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ebae4e2d4d0ea71c28228c5e0dbb8f23_20231225162831.pdf), [视频](https://www.bilibili.com/video/BV1ag4y1C7DD) | +| 游戏 | 腾讯互娱 | Meetup | 消灭盲点!腾讯游戏真·全栈观测实践 | [文章](https://mp.weixin.qq.com/s/vzRebv7TMrrRi8TUV9qj5A), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/580f8117457f0e2bbc2f3818f7d42300_20231225162841.pdf), [视频](https://www.bilibili.com/video/BV1ku4y1K7PF) | # 2023 -| 行业 | 用户 | 来源 | 标题 | 链接 | -| ------ | -------- | ------ | ------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| 新制造 | 某智能车企 | 直播 | 【可观测性实战】快速定位 K8s CNI 端口冲突问题 | [文章](https://mp.weixin.qq.com/s/Nb0FNSnYPkHC68Adv8QaRw), [视频](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a7570a4b46c4796f07572f3b7af00ddd_20230815170039.pdf) | -| 新制造 | 某智能车企 | 直播 | 【可观测性实战】快速定位云服务时延瓶颈 | [文章](https://mp.weixin.qq.com/s/Ex7o_n4dhZ4VgkPFYGCVFQ), [视频](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a7570a4b46c4796f07572f3b7af00ddd_20230815170039.pdf) | -| 物流 | 某物流企业 | 直播 | 【可观测性实战】快速定位 K8s 应用的时延瓶颈 | [文章](https://mp.weixin.qq.com/s/fzjbR8rlIOLd1eH0XDvM_w), [视频](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a7570a4b46c4796f07572f3b7af00ddd_20230815170039.pdf) | -| 电信 | 中国移动 | Meetup | 中国移动磐基 PaaS 平台基于 eBPF 的应用可观测性建设实践 | [文章](https://mp.weixin.qq.com/s/ACS4AXFUk0uCXAsVTBi2SQ) | -| 社区 | 郑志聪 | Meetup | DeepFlow 扩展协议解析实践 | [文章](https://mp.weixin.qq.com/s/GvUwamT-1VYHZQW34JBdow), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/50259d1f763207ff241a31b17231b871_20231201173751.pdf), [视频](https://www.bilibili.com/video/BV1pc411q7WH) | -| 电商 | 政采云 | Meetup | 政采云可观测性建设实践 | [文章](https://mp.weixin.qq.com/s/P_r1LQ3HerYNBYPZPClc2g), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/7698944121a1ce331c35428be49c2975_20230921103323.pdf), [视频](https://www.bilibili.com/video/BV1Sw411e7zC) | -| 电商 | 微拍堂 | Meetup | 微拍堂基于 DeepFlow 建设零侵扰的可观测平台 | [文章](https://mp.weixin.qq.com/s/P1tsmFW_9poIScxXCdOlLg), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ab5c0568c000db0d0669c8c6a59c3551_20230921103335.pdf), [视频](https://www.bilibili.com/video/BV1zH4y1S7zG) | -| 银行 | 光大银行 | 文章 | 浅谈分布式系统的性能调优 - Overlay 层数据包分析 | [文章](https://mp.weixin.qq.com/s/aXwH6IIjCwZYHHqtqP2NSQ) | -| 银行 | 民生银行 | 文章 | 云原生可观测性解决方案助力民生银行 IT 系统安全运维 | [文章](https://mp.weixin.qq.com/s/rcCSDZfauhDdRD32hf5oxw) | -| 游戏 | 趣丸科技 | 文章 | 完整指南:如何编译、打包和部署二次开发的 DeepFlow | [文章](https://mp.weixin.qq.com/s/-jWYq2rTRaTueuN0sAb3lA) | -| 社区 | Luga Lee | 文章 | 一文读懂基于 eBPF 自动化可观测平台 - DeepFlow | [文章](https://mp.weixin.qq.com/s/vkHsvoxJ6Ep-githtJAv7g) | -| 游戏 | 腾讯互娱 | Meetup | DeepFlow 在腾讯蓝鲸观测平台中的探索与实践 | [文章](https://www.infoq.cn/article/raua40qhu5ejhmqb0mf3), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/1de79730a61f2f03dce9890862733cf4_20231031154518.pdf), [视频](https://www.bilibili.com/video/BV1o14y1S7iy) | -| 互联网 | 小米集团 | Meetup | DeepFlow 在小米落地现状以及挑战 | [文章](https://mp.weixin.qq.com/s/0WMIdy1SoTYRTkU2e-PprQ), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a1ee4bcf5678dbd276353f4b59f4aeff_20231031154555.pdf), [视频](https://www.bilibili.com/video/BV12u411h7bn) | -| 软件 | 灵雀云 | Meetup | 记一次持续三个月的 K8s DNS 排障过程 | [文章](https://mp.weixin.qq.com/s/dDfckiTaALmFYHL6Tes_SA), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ff69a942735788d654ba3b7d5acc24c6_20231031154454.pdf), [视频](https://www.bilibili.com/video/BV13X4y147UN) | +| 行业 | 用户 | 来源 | 标题 | 链接 | +| ------ | ---------- | ------ | ------------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 新制造 | 某智能车企 | 直播 | 【可观测性实战】快速定位 K8s CNI 端口冲突问题 | [文章](https://mp.weixin.qq.com/s/Nb0FNSnYPkHC68Adv8QaRw), [视频](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a7570a4b46c4796f07572f3b7af00ddd_20230815170039.pdf) | +| 新制造 | 某智能车企 | 直播 | 【可观测性实战】快速定位云服务时延瓶颈 | [文章](https://mp.weixin.qq.com/s/Ex7o_n4dhZ4VgkPFYGCVFQ), [视频](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a7570a4b46c4796f07572f3b7af00ddd_20230815170039.pdf) | +| 物流 | 某物流企业 | 直播 | 【可观测性实战】快速定位 K8s 应用的时延瓶颈 | [文章](https://mp.weixin.qq.com/s/fzjbR8rlIOLd1eH0XDvM_w), [视频](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a7570a4b46c4796f07572f3b7af00ddd_20230815170039.pdf) | +| 电信 | 中国移动 | Meetup | 中国移动磐基 PaaS 平台基于 eBPF 的应用可观测性建设实践 | [文章](https://mp.weixin.qq.com/s/ACS4AXFUk0uCXAsVTBi2SQ) | +| 社区 | 郑志聪 | Meetup | DeepFlow 扩展协议解析实践 | [文章](https://mp.weixin.qq.com/s/GvUwamT-1VYHZQW34JBdow), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/50259d1f763207ff241a31b17231b871_20231201173751.pdf), [视频](https://www.bilibili.com/video/BV1pc411q7WH) | +| 电商 | 政采云 | Meetup | 政采云可观测性建设实践 | [文章](https://mp.weixin.qq.com/s/P_r1LQ3HerYNBYPZPClc2g), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/7698944121a1ce331c35428be49c2975_20230921103323.pdf), [视频](https://www.bilibili.com/video/BV1Sw411e7zC) | +| 电商 | 微拍堂 | Meetup | 微拍堂基于 DeepFlow 建设零侵扰的可观测平台 | [文章](https://mp.weixin.qq.com/s/P1tsmFW_9poIScxXCdOlLg), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ab5c0568c000db0d0669c8c6a59c3551_20230921103335.pdf), [视频](https://www.bilibili.com/video/BV1zH4y1S7zG) | +| 银行 | 光大银行 | 文章 | 浅谈分布式系统的性能调优 - Overlay 层数据包分析 | [文章](https://mp.weixin.qq.com/s/aXwH6IIjCwZYHHqtqP2NSQ) | +| 银行 | 民生银行 | 文章 | 云原生可观测性解决方案助力民生银行 IT 系统安全运维 | [文章](https://mp.weixin.qq.com/s/rcCSDZfauhDdRD32hf5oxw) | +| 游戏 | 趣丸科技 | 文章 | 完整指南:如何编译、打包和部署二次开发的 DeepFlow | [文章](https://mp.weixin.qq.com/s/-jWYq2rTRaTueuN0sAb3lA) | +| 社区 | Luga Lee | 文章 | 一文读懂基于 eBPF 自动化可观测平台 - DeepFlow | [文章](https://mp.weixin.qq.com/s/vkHsvoxJ6Ep-githtJAv7g) | +| 游戏 | 腾讯互娱 | Meetup | DeepFlow 在腾讯蓝鲸观测平台中的探索与实践 | [文章](https://www.infoq.cn/article/raua40qhu5ejhmqb0mf3), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/1de79730a61f2f03dce9890862733cf4_20231031154518.pdf), [视频](https://www.bilibili.com/video/BV1o14y1S7iy) | +| 互联网 | 小米集团 | Meetup | DeepFlow 在小米落地现状以及挑战 | [文章](https://mp.weixin.qq.com/s/0WMIdy1SoTYRTkU2e-PprQ), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a1ee4bcf5678dbd276353f4b59f4aeff_20231031154555.pdf), [视频](https://www.bilibili.com/video/BV12u411h7bn) | +| 软件 | 灵雀云 | Meetup | 记一次持续三个月的 K8s DNS 排障过程 | [文章](https://mp.weixin.qq.com/s/dDfckiTaALmFYHL6Tes_SA), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ff69a942735788d654ba3b7d5acc24c6_20231031154454.pdf), [视频](https://www.bilibili.com/video/BV13X4y147UN) | # 2022 diff --git a/docs/zh/02-ce-install/01-overview.md b/docs/zh/02-ce-install/01-overview.md index 19be7999..f04618ee 100644 --- a/docs/zh/02-ce-install/01-overview.md +++ b/docs/zh/02-ce-install/01-overview.md @@ -29,7 +29,7 @@ permalink: /ce-install/overview DeepFlow 中的 eBPF 能力(AutoTracing、AutoProfiling)对内核版本的要求如下: | 体系架构 | 发行版 | 内核版本 | kprobe [1] | Golang uprobe | OpenSSL uprobe | perf | -| -------- | ------ | ------- | ---------- | ------------- | -------------- | ---- | +| -------- | ------ | ------- | ---------- | ------------- | -------------- | ---- | | X86 | CentOS 7.9 | 3.10.0-940+ **[2]** | Y | Y **[3]** | Y **[3]** | Y | | | RedHat 7.6 | 3.10.0-940+ **[2]** | Y | Y **[3]** | Y **[3]** | Y | | | \* | 4.14 **[4]** | Y | Y **[3]** | | Y | diff --git a/docs/zh/02-ce-install/02-all-in-one.md b/docs/zh/02-ce-install/02-all-in-one.md index 959c0a6a..986f0f71 100644 --- a/docs/zh/02-ce-install/02-all-in-one.md +++ b/docs/zh/02-ce-install/02-all-in-one.md @@ -108,7 +108,7 @@ Grafana auth: admin:deepflow 我们不推荐使用 Docker 部署 DeepFlow Server 端,具体原因如下: -1. Server 端依赖 K8s 的 [lease](https://kubernetes.io/zh-cn/docs/concepts/architecture/leases/) 进行选主,通过多副本实现高可用性。而 Docker 环境缺乏 K8s 的这一机制,导致 Server 端仅能以单副本模式运行。在 Agent 节点数量较多或数据采集量较大的场景下,单副本实例可能因资源瓶颈而无法承载高并发的数据量。 +1. Server 端依赖 K8s 的 [lease](https://kubernetes.io/zh-cn/docs/concepts/architecture/leases/) 进行选主,通过多副本实现高可用性。而 Docker 环境缺乏 K8s 的这一机制,导致 Server 端仅能以单副本模式运行。在 Agent 节点数量较多或数据采集量较大的场景下,单副本实例可能因资源瓶颈而无法承载高并发的数据量。 2. 当 Server 端以单副本形式部署时,配套的 ClickHouse 只能采用单切片形式部署,否则会导致数据写入不均衡,这在一定程度上限制了数据的查询速度。 ## 准备工作 @@ -169,6 +169,7 @@ docker compose -f deepflow-docker-compose/docker-compose.yaml up -d 通过 docker compose 部署后,将浏览器指向 `http://<$NODE_IP_FOR_DEEPFLOW>:3000` 即可登录 Grafana 控制台 默认凭据: + - 用户名:admin - 密码:deepflow diff --git a/docs/zh/02-ce-install/08-serverless-pod.md b/docs/zh/02-ce-install/08-serverless-pod.md index 7a776999..f95fc5c8 100644 --- a/docs/zh/02-ce-install/08-serverless-pod.md +++ b/docs/zh/02-ce-install/08-serverless-pod.md @@ -42,6 +42,7 @@ helm install deepflow-agent -n deepflow deepflow/deepflow-agent --version 6.6.01 ``` 上述命令将会部署两组 deepflow-agent: + - watcher:一个 deepflow-agent 的 deployment,用于同步 K8s 资源。 - 部署时会自动注入 `K8S_WATCH_POLICY=watch-only` 的环境变量,此时 deepflow-agent 仅会同步 K8s 资源,不会采集可观测性数据。 - daemonset:在每个 serverless pod 中以 sidecar 的方式注入 deepflow-agent,用于采集可观测性数据。 diff --git a/docs/zh/02-ce-install/09-ai-agent.md b/docs/zh/02-ce-install/09-ai-agent.md index cfb04c3a..4f60e29d 100644 --- a/docs/zh/02-ce-install/09-ai-agent.md +++ b/docs/zh/02-ce-install/09-ai-agent.md @@ -11,20 +11,21 @@ DeepFlow 部署时,默认不会启用 AI 组件,需手动在 `values-custom. stella-agent-ce: enabled: true replicas: 1 - hostNetwork: "false" + hostNetwork: 'false' dnsPolicy: ClusterFirst imagePullSecrets: [] - nameOverride: "" - fullnameOverride: "" + nameOverride: '' + fullnameOverride: '' podAnnotations: {} image: - repository: "{{ .Values.global.image.repository }}/deepflowio-stella-agent-ce" + repository: '{{ .Values.global.image.repository }}/deepflowio-stella-agent-ce' pullPolicy: Always # Overrides the image tag whose default is the chart appVersion. tag: latest - podSecurityContext: {} + podSecurityContext: + {} # fsGroup: 2000 securityContext: @@ -40,19 +41,19 @@ stella-agent-ce: ## Configuration for ClickHouse service annotations: {} labels: {} - clusterIP: "" + clusterIP: '' ## Port for ClickHouse Service to listen on ports: - - name: tcp - port: 20831 - targetPort: 20831 - nodePort: - protocol: TCP + - name: tcp + port: 20831 + targetPort: 20831 + nodePort: + protocol: TCP # Additional ports to open for server service additionalPorts: [] externalIPs: [] - loadBalancerIP: "" + loadBalancerIP: '' loadBalancerSourceRanges: [] ## Denotes if this Service desires to route external traffic to node-local or cluster-wide endpoints @@ -81,18 +82,19 @@ stella-agent-ce: df-llm-agent.yaml: daemon: true api_timeout: 500 - sql_show: "false" - log_file: "/var/log/df-llm-agent.log" - log_level: "info" - instance_path: "/root/df-llm-agent" + sql_show: 'false' + log_file: '/var/log/df-llm-agent.log' + log_level: 'info' + instance_path: '/root/df-llm-agent' mysql: - host: "{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.ip}}{{ else }}{{ $.Release.Name }}-mysql{{end}}" - port: "{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.port}}{{ else }}30130{{end}}" - user_name: "{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.username}}{{ else }}root{{end}}" - user_password: "{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.password}}{{ else }}{{ .Values.global.password.mysql }}{{end}}" - database: "deepflow_llm" - - resources: {} + host: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.ip}}{{ else }}{{ $.Release.Name }}-mysql{{end}}' + port: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.port}}{{ else }}30130{{end}}' + user_name: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.username}}{{ else }}root{{end}}' + user_password: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.password}}{{ else }}{{ .Values.global.password.mysql }}{{end}}' + database: 'deepflow_llm' + + resources: + {} # limits: # cpu: 100m # memory: 128Mi @@ -108,7 +110,8 @@ stella-agent-ce: podAntiAffinityTermLabelSelector: [] podAffinityLabelSelector: [] podAffinityTermLabelSelector: [] - nodeAffinityLabelSelector: [] + nodeAffinityLabelSelector: + [] # - matchExpressions: # - key: kubernetes.io/hostname # operator: In diff --git a/docs/zh/04-best-practice/03-special-environment-deployment.md b/docs/zh/04-best-practice/03-special-environment-deployment.md index a7970292..d3dc4729 100644 --- a/docs/zh/04-best-practice/03-special-environment-deployment.md +++ b/docs/zh/04-best-practice/03-special-environment-deployment.md @@ -122,6 +122,7 @@ K8s 使用 macvlan CNI 时,在 rootns 下只能看到所有 POD 共用的一 参考[文档](../configuration/agent/#inputs.cbpf.af_packet.inner_interface_capture_enabled),开启 deepflow-agent 的 `inputs.cbpf.af_packet.inner_interface_capture_enabled`,可采集 PodNS 中的网卡流量。 注意需要同时调整如下配置: + - `inputs.cbpf.af_packet.tunning.ring_blocks_enabled`:使得能够让 AF_PACKET 的内存消耗可调整。 - `inputs.cbpf.af_packet.tunning.ring_blocks`:使得能够精简所有 AF_PACKET 的总体内存消耗。 - `inputs.cbpf.af_packet.inner_interface_regex`:使得能够正确匹配 PodNS 内部的网卡名称。 @@ -138,7 +139,6 @@ K8s 使用 macvlan CNI 时,在 rootns 下只能看到所有 POD 共用的一 唯一需要注意的是,Agent 的 tap_interface_regex 只需配置为 Node NIC 列表。 - # K8s 运行 Agent 权限受限 ## 无 K8s Daemonset 部署权限 diff --git a/docs/zh/04-best-practice/07-storage-engine-use-byconity.md b/docs/zh/04-best-practice/07-storage-engine-use-byconity.md index ef2f9270..f821840a 100644 --- a/docs/zh/04-best-practice/07-storage-engine-use-byconity.md +++ b/docs/zh/04-best-practice/07-storage-engine-use-byconity.md @@ -11,13 +11,15 @@ permalink: /best-practice/storage-engine-use-byconity/ ::: tip ByConity 共有 17 个 Pod,其中 9 个 Pod 的 Request 和 Limit 为 1.1C 1280M,1 个 Pod 的 Request 和 Limit 为 1C 1G,1 个 Pod 的 Request 和 Limit 为 1C 512M。组件 `byconity-server`、`vw-default` 和 `vw-writer` 的本地 Disk Cache 可通过 `lru_max_size` 配置修改,日志数据存储上限可通过 `size`、`count` 配置修改。 -资源需求: +资源需求: + - CPU: 建议 Kubernetes 集群至少剩余 12C 可分配资源,实际会消耗更高的资源。 - 内存: 建议 Kubernetes 集群至少剩余 14G 可分配资源,实际会消耗更高的资源。 - 磁盘: 建议每个数据节点磁盘容量超过 180G,其中本地 Disk Cache `byconity-server`、`vw-default` 和 `vw-writer` 各 40G,日志数据 `byconity-server`、`vw-default` 和 `vw-writer` 各 20G。 -::: + ::: ## 部署参数 + ByConity 默认对接对象存储,环境要求可参考官方说明。部署时在自定义 values-custom.yaml 文件中添加 byconity 配置即可。 注:本次配置以阿里云 OSS 为例,须将示例中 `endpoint`,`region`,`bucket`,`path`,`ak_id`,`ak_secret` 修改为对象存储的正确参数,并建议将 `byconity-server`、`vw-default` 和 `vw-writer` 副本数量调整至与 `deepflow-server` 或节点数量相同。 @@ -30,22 +32,22 @@ clickhouse: byconity: enabled: true - nameOverride: "" - fullnameOverride: "" + nameOverride: '' + fullnameOverride: '' image: - repository: "{{ .Values.global.image.repository }}/byconity" + repository: '{{ .Values.global.image.repository }}/byconity' tag: 1.0.0 imagePullPolicy: IfNotPresent fdbShell: image: - repository: "{{ .Values.global.image.repository }}" + repository: '{{ .Values.global.image.repository }}' byconity: configOverwrite: storage_configuration: cnch_default_policy: cnch_default_s3 disks: - server_s3_disk_0: # FIXME + server_s3_disk_0: # FIXME path: byconity0 endpoint: https://oss-cn-beijing-internal.aliyuncs.com region: cn-beijing @@ -61,7 +63,7 @@ byconity: bytes3: default: server_s3_disk_0 disk: server_s3_disk_0 - + ports: tcp: 9000 http: 8123 @@ -74,7 +76,7 @@ byconity: usersOverwrite: users: default: - password: "" + password: '' probe: password: probe profiles: @@ -83,17 +85,17 @@ byconity: enable_multiple_tables_for_cnch_parts: 1 server: - replicas: 1 # FIXME - image: "" - podAnnotations: { } - resources: { } + replicas: 1 # FIXME + image: '' + podAnnotations: {} + resources: {} hostNetwork: false - nodeSelector: { } - tolerations: [ ] + nodeSelector: {} + tolerations: [] affinity: nodeAffinity: {} - imagePullSecrets: [ ] - securityContext: { } + imagePullSecrets: [] + securityContext: {} storage: localDisk: pvcSpec: @@ -121,17 +123,17 @@ byconity: tso: replicas: 1 - image: "" - podAnnotations: { } - resources: { } + image: '' + podAnnotations: {} + resources: {} hostNetwork: false - nodeSelector: { } - tolerations: [ ] + nodeSelector: {} + tolerations: [] affinity: {} - imagePullSecrets: [ ] - securityContext: { } - configOverwrite: { } - additionalVolumes: { } + imagePullSecrets: [] + securityContext: {} + configOverwrite: {} + additionalVolumes: {} storage: localDisk: pvcSpec: @@ -140,7 +142,7 @@ byconity: resources: requests: storage: 10Gi - storageClassName: openebs-hostpath # FIXME: replace to your storageClassName + storageClassName: openebs-hostpath # FIXME: replace to your storageClassName log: pvcSpec: accessModes: @@ -148,48 +150,48 @@ byconity: resources: requests: storage: 10Gi - storageClassName: openebs-hostpath # FIXME: replace to your storageClassName + storageClassName: openebs-hostpath # FIXME: replace to your storageClassName daemonManager: replicas: 1 # Please keep single instance now, daemon manager HA is WIP - image: "" - podAnnotations: { } - resources: { } + image: '' + podAnnotations: {} + resources: {} hostNetwork: false - nodeSelector: { } - tolerations: [ ] + nodeSelector: {} + tolerations: [] affinity: {} - imagePullSecrets: [ ] - securityContext: { } - configOverwrite: { } + imagePullSecrets: [] + securityContext: {} + configOverwrite: {} resourceManager: replicas: 1 - image: "" - podAnnotations: { } - resources: { } + image: '' + podAnnotations: {} + resources: {} hostNetwork: false - nodeSelector: { } - tolerations: [ ] + nodeSelector: {} + tolerations: [] affinity: {} - imagePullSecrets: [ ] - securityContext: { } - configOverwrite: { } + imagePullSecrets: [] + securityContext: {} + configOverwrite: {} defaultWorker: &defaultWorker replicas: 1 - image: "" - podAnnotations: { } - resources: { } + image: '' + podAnnotations: {} + resources: {} hostNetwork: false - nodeSelector: { } - tolerations: [ ] + nodeSelector: {} + tolerations: [] affinity: {} - imagePullSecrets: [ ] - securityContext: { } + imagePullSecrets: [] + securityContext: {} livenessProbe: exec: - command: [ "/opt/byconity/scripts/lifecycle/liveness" ] + command: ['/opt/byconity/scripts/lifecycle/liveness'] failureThreshold: 6 initialDelaySeconds: 5 periodSeconds: 10 @@ -197,7 +199,7 @@ byconity: timeoutSeconds: 20 readinessProbe: exec: - command: [ "/opt/byconity/scripts/lifecycle/readiness" ] + command: ['/opt/byconity/scripts/lifecycle/readiness'] failureThreshold: 5 initialDelaySeconds: 10 periodSeconds: 10 @@ -231,49 +233,49 @@ byconity: virtualWarehouses: - <<: *defaultWorker name: vw_default - replicas: 1 # FIXME + replicas: 1 # FIXME - <<: *defaultWorker name: vw_write - replicas: 1 # FIXME + replicas: 1 # FIXME commonEnvs: - name: MY_POD_NAMESPACE valueFrom: fieldRef: - fieldPath: "metadata.namespace" + fieldPath: 'metadata.namespace' - name: MY_POD_NAME valueFrom: fieldRef: - fieldPath: "metadata.name" + fieldPath: 'metadata.name' - name: MY_UID valueFrom: fieldRef: apiVersion: v1 - fieldPath: "metadata.uid" + fieldPath: 'metadata.uid' - name: MY_POD_IP valueFrom: fieldRef: - fieldPath: "status.podIP" + fieldPath: 'status.podIP' - name: MY_HOST_IP valueFrom: fieldRef: # fieldPath: "status.hostIP" - fieldPath: "status.podIP" + fieldPath: 'status.podIP' - name: CONSUL_HTTP_HOST valueFrom: fieldRef: - fieldPath: "status.hostIP" + fieldPath: 'status.hostIP' - additionalEnvs: [ ] + additionalEnvs: [] additionalVolumes: - volumes: [ ] - volumeMounts: [ ] + volumes: [] + volumeMounts: [] - postStart: "" - preStop: "" - livenessProbe: "" - readinessProbe: "" + postStart: '' + preStop: '' + livenessProbe: '' + readinessProbe: '' ingress: enabled: false @@ -286,14 +288,14 @@ byconity: clusterSpec: mainContainer: imageConfigs: - - version: 7.1.15 - baseImage: "{{ .Values.global.image.repository }}/foundationdb" - tag: 7.1.15 + - version: 7.1.15 + baseImage: '{{ .Values.global.image.repository }}/foundationdb' + tag: 7.1.15 sidecarContainer: imageConfigs: - - version: 7.1.15 - baseImage: "{{ .Values.global.image.repository }}/foundationdb-kubernetes-sidecar" - tag: 7.1.15-1 + - version: 7.1.15 + baseImage: '{{ .Values.global.image.repository }}/foundationdb-kubernetes-sidecar' + tag: 7.1.15-1 processCounts: stateless: 3 log: 3 @@ -318,25 +320,25 @@ byconity: memory: 512Mi affinity: {} image: - repository: "{{ .Values.global.image.repository }}/fdb-kubernetes-operator" + repository: '{{ .Values.global.image.repository }}/fdb-kubernetes-operator' tag: v1.9.0 pullPolicy: IfNotPresent initContainerImage: - repository: "{{ $.Values.global.image.repository }}/foundationdb-kubernetes-sidecar" + repository: '{{ $.Values.global.image.repository }}/foundationdb-kubernetes-sidecar' initContainers: 6.2: image: - repository: "{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar" + repository: '{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar' tag: 6.2.30-1 pullPolicy: IfNotPresent 6.3: image: - repository: "{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar" + repository: '{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar' tag: 6.3.23-1 pullPolicy: IfNotPresent 7.1: image: - repository: "{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar" + repository: '{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar' tag: 7.1.15-1 pullPolicy: IfNotPresent hdfs: @@ -354,17 +356,17 @@ helm install deepflow -n deepflow -f values-custom.yaml deepflow/deepflow - ByConity 只支持 AMD64 架构。 - 如果出现部分 `byconity-fdb-storage` Pod 启动失败的情况,请调整内核参数: - ```bash - sudo sysctl -w fs.inotify.max_user_watches=2099999999 - sudo sysctl -w fs.inotify.max_user_instances=2099999999 - sudo sysctl -w fs.inotify.max_queued_events=2099999999 - ``` + ```bash + sudo sysctl -w fs.inotify.max_user_watches=2099999999 + sudo sysctl -w fs.inotify.max_user_instances=2099999999 + sudo sysctl -w fs.inotify.max_queued_events=2099999999 + ``` - Byconity 依赖于 FoundationDB 集群(简称 FDB),该集群用于存储 Byconity 的元数据。若对 FDB 集群进行删除或重建操作,将会导致 FDB 数据的丢失,进而引发 Byconity 数据的丢失。因此,在卸载 Byconity 的过程中,不会删除 FDB 组件。若确实需要删除该组件,请执行相应的删除操作: - ```bash - kubectl delete FoundationDBCluster --all -n deepflow - ``` + ```bash + kubectl delete FoundationDBCluster --all -n deepflow + ``` - 使用私有仓库导致 FDB 部分组件无法拉取镜像情况,可以使用如下命令解决: - ```bash - kubectl patch serviceaccount default -p '{"imagePullSecrets": [{"name": "myregistrykey"}]}' -n deepflow - kubectl delete pod -n deepflow -l foundationdb.org/fdb-cluster-name=deepflow-byconity-fdb - ``` + ```bash + kubectl patch serviceaccount default -p '{"imagePullSecrets": [{"name": "myregistrykey"}]}' -n deepflow + kubectl delete pod -n deepflow -l foundationdb.org/fdb-cluster-name=deepflow-byconity-fdb + ``` diff --git a/docs/zh/05-features/01-l7-protocols/01-overview.md b/docs/zh/05-features/01-l7-protocols/01-overview.md index 0f9550cf..f4e1e479 100644 --- a/docs/zh/05-features/01-l7-protocols/01-overview.md +++ b/docs/zh/05-features/01-l7-protocols/01-overview.md @@ -6,6 +6,7 @@ permalink: /features/l7-protocols/overview # 支持的应用协议 为了降低资源开销并避免误识别,agent 默认仅会解析如下应用协议: + - HTTP、HTTP2/gRPC、MySQL、Redis、Kafka、DNS、TLS。 如需开启其他应用协议的解析,请配置 agent 的 `l7-protocol-enabled`。支持解析的所有应用协议如下所示: diff --git a/docs/zh/05-features/01-l7-protocols/03-rpc.md b/docs/zh/05-features/01-l7-protocols/03-rpc.md index afe2a08d..aae6a7cb 100644 --- a/docs/zh/05-features/01-l7-protocols/03-rpc.md +++ b/docs/zh/05-features/01-l7-protocols/03-rpc.md @@ -218,22 +218,22 @@ permalink: /features/l7-protocols/rpc **Tag 字段映射表格,以下表格只包含存在映射关系的字段** -| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | -| ----- | ------------------ | ------------ | ------------------------ | -------------------------------------------------------------------------------- | -------------------------------------------- | -| Req. | version | 协议版本 | tars_version | -- | -- | -| | request_type | 请求类型 | request.method_name | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | request.service_name | -- | -- | -| | request_id | 请求 ID | request_id | -- | -- | -| | endpoint | 端点 | service_name/method_name | -- | -- | -| Resp. | response_code | 响应码 | -- | Status Code | -- | -| | response_status | 响应状态 | -- | response.status | 0 正常,-10 至 -12 客户端异常,其余服务端异常 | -| | response_exception | 响应异常 | -- | 参考返回码描述,详见 [iRet Code](https://doc.tarsyun.com/#/base/tars-protocol.md) | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | +| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | +| ----- | ------------------ | ------------ | ------------------------ | --------------------------------------------------------------------------------- | --------------------------------------------- | +| Req. | version | 协议版本 | tars_version | -- | -- | +| | request_type | 请求类型 | request.method_name | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | request.service_name | -- | -- | +| | request_id | 请求 ID | request_id | -- | -- | +| | endpoint | 端点 | service_name/method_name | -- | -- | +| Resp. | response_code | 响应码 | -- | Status Code | -- | +| | response_status | 响应状态 | -- | response.status | 0 正常,-10 至 -12 客户端异常,其余服务端异常 | +| | response_exception | 响应异常 | -- | 参考返回码描述,详见 [iRet Code](https://doc.tarsyun.com/#/base/tars-protocol.md) | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | **Metrics 字段映射表格,以下表格只包含存在映射关系的字段** diff --git a/docs/zh/05-features/01-l7-protocols/05-nosql.md b/docs/zh/05-features/01-l7-protocols/05-nosql.md index 2f4e891a..b5d1f81d 100644 --- a/docs/zh/05-features/01-l7-protocols/05-nosql.md +++ b/docs/zh/05-features/01-l7-protocols/05-nosql.md @@ -77,28 +77,27 @@ permalink: /features/l7-protocols/nosql | client_error_ratio | 客户端异常比例 | -- | -- | 客户端异常 / 响应 | | server_error_ratio | 服务端异常比例 | -- | -- | 服务端异常 / 响应 | - # Memcached 通过解析 [Memcached](https://github.com/memcached/memcached/blob/master/doc/protocol.txt) 协议,将 Memcached Request / Response 的字段映射到 l7_flow_log 对应字段中,映射关系如下表: **Tag 字段映射表格,以下表格只包含存在映射关系的字段** -| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | +| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | | ----- | ------------------ | ------------ | ------------------- | --------------------------- | -------------- | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_type | 请求类型 | Payload 首个单词 | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | Payload 首行(\r\n)| -- | -- | -| | request_id | 请求 ID | -- | -- | -- | -| | endpoint | 端点 | -- | -- | -- | -| Resp. | response_code | 响应码 | -- | -- | -- | -| | response_status | 响应状态 | -- | Payload 首个单词 | -- | -| | response_exception | 响应异常 | -- | 异常时 Payload 首行错误信息 | -- | -| | response_result | 响应结果 | -- | Payload 首行 | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_type | 请求类型 | Payload 首个单词 | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | Payload 首行(\r\n)| -- | -- | +| | request_id | 请求 ID | -- | -- | -- | +| | endpoint | 端点 | -- | -- | -- | +| Resp. | response_code | 响应码 | -- | -- | -- | +| | response_status | 响应状态 | -- | Payload 首个单词 | -- | +| | response_exception | 响应异常 | -- | 异常时 Payload 首行错误信息 | -- | +| | response_result | 响应结果 | -- | Payload 首行 | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | **Metrics 字段映射表格,以下表格只包含存在映射关系的字段** diff --git a/docs/zh/05-features/01-l7-protocols/06-mq.md b/docs/zh/05-features/01-l7-protocols/06-mq.md index 70082d70..97b62a29 100644 --- a/docs/zh/05-features/01-l7-protocols/06-mq.md +++ b/docs/zh/05-features/01-l7-protocols/06-mq.md @@ -310,22 +310,22 @@ permalink: /features/l7-protocols/mq **Tag 字段映射表格,以下表格只包含存在映射关系的字段** -| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | -| ----- | ------------------ | ------------ | ---------------------------------------- | --------------- | --------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | version | -- | -- | -| | request_type | 请求类型 | code | -- | -- | -| | request_domain | 请求域名 | extFields:producerGroup \| consumerGroup | -- | 主要针对 SEND 和 PULL ,其他消息暂未完整补充 | -| | request_resource | 请求资源 | extFields:topic | -- | 主要针对 SEND 和 PULL ,其他消息暂未完整补充 | -| | request_id | 请求 ID | opaque | -- | -- | -| | endpoint | 端点 | extFields:topic & queueId | -- | -- | -| Resp. | response_code | 响应码 | -- | code | -- | -| | response_status | 响应状态 | -- | code | 为 0 代表正常,非 0 代表各种异常 | -| | response_exception | 响应异常 | -- | remark | -- | -| | response_result | 响应结果 | -- | body | JSON 序列化格式时对应所有 JSON 字符串数据, ROCKETMQ 格式时对应其中的 body 字符串 | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8| 从 extFields 或者 bodyData 的 properties 字段中提取 | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8| 从 extFields 或者 bodyData 的 properties 字段中提取 | -| | x_request_id | X-Request-ID | UNIQ_KEY | KEY | 从 extFields 或者 bodyData 的 properties 字段中提取 | -| Misc. | -- | -- | -- | -- | -- | +| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | +| ----- | ------------------ | ------------ | ---------------------------------------- | ---------------- | --------------------------------------------------------------------------------- | +| Req. | version | 协议版本 | version | -- | -- | +| | request_type | 请求类型 | code | -- | -- | +| | request_domain | 请求域名 | extFields:producerGroup \| consumerGroup | -- | 主要针对 SEND 和 PULL ,其他消息暂未完整补充 | +| | request_resource | 请求资源 | extFields:topic | -- | 主要针对 SEND 和 PULL ,其他消息暂未完整补充 | +| | request_id | 请求 ID | opaque | -- | -- | +| | endpoint | 端点 | extFields:topic & queueId | -- | -- | +| Resp. | response_code | 响应码 | -- | code | -- | +| | response_status | 响应状态 | -- | code | 为 0 代表正常,非 0 代表各种异常 | +| | response_exception | 响应异常 | -- | remark | -- | +| | response_result | 响应结果 | -- | body | JSON 序列化格式时对应所有 JSON 字符串数据, ROCKETMQ 格式时对应其中的 body 字符串 | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | 从 extFields 或者 bodyData 的 properties 字段中提取 | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | 从 extFields 或者 bodyData 的 properties 字段中提取 | +| | x_request_id | X-Request-ID | UNIQ_KEY | KEY | 从 extFields 或者 bodyData 的 properties 字段中提取 | +| Misc. | -- | -- | -- | -- | -- | - RocketMQ 协议中,flag 字段的 bit0 会标识是请求还是响应,而 bit1 会标识是否为单向请求(不需要响应) - 对于 SEND_MESSAGE 、 PULL_MESSAGE 这类非单向请求和对应的响应会按照 opaque 一对一聚合为一个 Session diff --git a/docs/zh/05-features/01-l7-protocols/07-network.md b/docs/zh/05-features/01-l7-protocols/07-network.md index 99fd1445..5f2bb52e 100644 --- a/docs/zh/05-features/01-l7-protocols/07-network.md +++ b/docs/zh/05-features/01-l7-protocols/07-network.md @@ -46,20 +46,20 @@ permalink: /features/l7-protocols/network **Tag 字段映射表格,以下表格只包含存在映射关系的字段** -| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | -| ----- | ------------------ | ------------ | -------------- | --------------- | --------------------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_resource | 请求资源 | Identifier | -- | -- | -| | request_id | 请求 ID | Sequence Number | -- | -- | -| | endpoint | 端点 | -- | -- | -- | -| Resp. | response_code | 响应码 | -- | -- | -- | -| | response_status | 响应状态 | -- | -- | 收到响应,则记录为正常;未收到响应则记录为超时 -| | response_exception | 响应异常 | -- | -- | -- -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | +| 类别 | 名称 | 中文 | Request Header | Response Header | 描述 | +| ----- | ------------------ | ------------ | --------------- | --------------- | ---------------------------------------------- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_resource | 请求资源 | Identifier | -- | -- | +| | request_id | 请求 ID | Sequence Number | -- | -- | +| | endpoint | 端点 | -- | -- | -- | +| Resp. | response_code | 响应码 | -- | -- | -- | +| | response_status | 响应状态 | -- | -- | 收到响应,则记录为正常;未收到响应则记录为超时 | +| | response_exception | 响应异常 | -- | -- | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | **Metrics 字段映射表格,以下表格只包含存在映射关系的字段** diff --git a/docs/zh/05-features/01-l7-protocols/08-otel.md b/docs/zh/05-features/01-l7-protocols/08-otel.md index a864ccd8..93d48b66 100644 --- a/docs/zh/05-features/01-l7-protocols/08-otel.md +++ b/docs/zh/05-features/01-l7-protocols/08-otel.md @@ -28,7 +28,7 @@ permalink: /features/l7-protocols/otel | service_name | 服务名称 | resource./span.attribute.service.name | -- | | service_instance_id | 服务实例 | resource./span.attribute.service.instance.id | -- | | endpoint | 端点 | span.name | -- | -| trace_id | TraceID | resource.attribute.sw8.trace_id/span.attribute.sw8.trace_id/span.trace_id | 优先级 resource.attribute.sw8.trace_id > span.attribute.sw8.trace_id > span.trace_id | +| trace_id | TraceID | resource.attribute.sw8.trace_id/span.attribute.sw8.trace_id/span.trace_id | 优先级 resource.attribute.sw8.trace_id > span.attribute.sw8.trace_id > span.trace_id | | span_id | SpanID | span.span_id/attribute.sw8.segment_id-attribute.sw8.span_id | 优先使用 attribute.sw8.segment_id-attribute.sw8.span_id | | parent_span_id | ParentSpanID | span.parent_span_id/attribute.sw8.segment_id-attribute.sw8.parent_span_id | 优先使用 attribute.sw8.segment_id-attribute.sw8.parent_span_id | | span_kind | Span 类型 | span.span_kind | -- | diff --git a/docs/zh/05-features/01-l7-protocols/09-skywalking.md b/docs/zh/05-features/01-l7-protocols/09-skywalking.md index 0359c0e3..bafdc4a4 100644 --- a/docs/zh/05-features/01-l7-protocols/09-skywalking.md +++ b/docs/zh/05-features/01-l7-protocols/09-skywalking.md @@ -7,30 +7,30 @@ permalink: /features/l7-protocols/skywalking **Tag 字段映射表格,以下表格只包含存在映射关系的字段** -| 名称 | 中文 | SkyWalking 数据结构 | 描述 | -|--------------------|--------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------------| -| start_time | 开始时间 | span.startTime | -- | -| end_time | 结束时间 | span.endTime | -- | -| protocol | 网络协议 | TCP | 固定枚举值 | -| attributes | 标签 | span.tags | -- | -| ip | IP 地址 | -- | 从上游 SkyWalking Agent 发送数据源获取 | -| l7_protocol | 应用协议 | span.tags.[db.type/http.scheme/db.system/rpc.system/messaging.system/messaging.protocol] | 如果有任何 `http.` 开头的 tags,则标记为 HTTP 协议,否则尝试从 span.tags 中获取 | -| l7_protocol_str | 应用协议 | span.tags.[db.type/http.scheme/db.system/rpc.system/messaging.system/messaging.protocol] | 先从 l7_protocol 中尝试转换为描述,如果无法转换则直接记录为 tag value | -| version | 协议版本 | span.tags.http.flavor | -- | -| type | 日志类型 | SESSION | 固定枚举值 | -| request_type | 请求类型 | span.tags.[http.method/cache.cmd/db.operation/rpc.method] | -- | -| request_domain | 请求域名 | span.tags.[http.host/db.connection_string] | -- | -| request_resource | 请求资源 | span.endpointName/span.tags.[http.target/db.statement/messaging.url/rpc.service/cache.key] | span.tags.http.target 如果存在则读取,不存在则从 http.url 截断取, 仅提取域名之后的调用信息 | -| request_id | 请求 ID | -- | | -| response_status | 响应状态 | 根据 response_code 尝试转换,若无法转换,获取 span.isError,若为 true: STATUS_SERVER_ERROR,否则 STATUS_SERVER_OK | -- | -| response_code | 响应码 | span.tags.[status_code/http.status_code/status.code] | 优先使用 span.tags.http.status_code | -| response_exception | 响应异常 | 根据 response_code 转换对应异常描述 | -- | -| app_service | 服务名称 | segment.service | -- | -| app_instance | 服务实例 | segment.serviceInstance | -- | -| endpoint | 端点 | span.operationName | -- | -| trace_id | TraceID | span.traceID | -- | -| span_id | SpanID | span.TraceSegmentID-span.spanID | -- | +| 名称 | 中文 | SkyWalking 数据结构 | 描述 | +| ------------------ | ------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------- | +| start_time | 开始时间 | span.startTime | -- | +| end_time | 结束时间 | span.endTime | -- | +| protocol | 网络协议 | TCP | 固定枚举值 | +| attributes | 标签 | span.tags | -- | +| ip | IP 地址 | -- | 从上游 SkyWalking Agent 发送数据源获取 | +| l7_protocol | 应用协议 | span.tags.[db.type/http.scheme/db.system/rpc.system/messaging.system/messaging.protocol] | 如果有任何 `http.` 开头的 tags,则标记为 HTTP 协议,否则尝试从 span.tags 中获取 | +| l7_protocol_str | 应用协议 | span.tags.[db.type/http.scheme/db.system/rpc.system/messaging.system/messaging.protocol] | 先从 l7_protocol 中尝试转换为描述,如果无法转换则直接记录为 tag value | +| version | 协议版本 | span.tags.http.flavor | -- | +| type | 日志类型 | SESSION | 固定枚举值 | +| request_type | 请求类型 | span.tags.[http.method/cache.cmd/db.operation/rpc.method] | -- | +| request_domain | 请求域名 | span.tags.[http.host/db.connection_string] | -- | +| request_resource | 请求资源 | span.endpointName/span.tags.[http.target/db.statement/messaging.url/rpc.service/cache.key] | span.tags.http.target 如果存在则读取,不存在则从 http.url 截断取, 仅提取域名之后的调用信息 | +| request_id | 请求 ID | -- | | +| response_status | 响应状态 | 根据 response_code 尝试转换,若无法转换,获取 span.isError,若为 true: STATUS_SERVER_ERROR,否则 STATUS_SERVER_OK | -- | +| response_code | 响应码 | span.tags.[status_code/http.status_code/status.code] | 优先使用 span.tags.http.status_code | +| response_exception | 响应异常 | 根据 response_code 转换对应异常描述 | -- | +| app_service | 服务名称 | segment.service | -- | +| app_instance | 服务实例 | segment.serviceInstance | -- | +| endpoint | 端点 | span.operationName | -- | +| trace_id | TraceID | span.traceID | -- | +| span_id | SpanID | span.TraceSegmentID-span.spanID | -- | | parent_span_id | ParentSpanID | segment.ID-span.parentSpanID/span.ref.parentTraceSegmentID-span.ref.parentSpanID | 优先 segment.ID-span.parentSpanID,若 span.parentSpanID = -1 则从 span.ref 中获取 ParentSpanID | -| span_kind | Span 类型 | span.spanType.Exit: SPAN_KIND_CLIENT, span.spanType.Entry: SPAN_KIND_SERVER, span.spanType.Local: SPAN_KIND_INTERNAL, span.spanType.Entry && span.spanLayer.MQ: SPAN_KIND_CONSUMER, span.spanType.Exit && span.spanLayer.MQ: SPAN_KIND_PRODUCER | -- | -| events | 事件 | -- | -- | -| observation_point | 观测点 | span.spanType.Exit: 客户端应用(C-APP), span.spanType.Entry: 服务端应用(S-APP), span.spanType.Local: 应用(APP) | -- | +| span_kind | Span 类型 | span.spanType.Exit: SPAN_KIND_CLIENT, span.spanType.Entry: SPAN_KIND_SERVER, span.spanType.Local: SPAN_KIND_INTERNAL, span.spanType.Entry && span.spanLayer.MQ: SPAN_KIND_CONSUMER, span.spanType.Exit && span.spanLayer.MQ: SPAN_KIND_PRODUCER | -- | +| events | 事件 | -- | -- | +| observation_point | 观测点 | span.spanType.Exit: 客户端应用(C-APP), span.spanType.Entry: 服务端应用(S-APP), span.spanType.Local: 应用(APP) | -- | diff --git a/docs/zh/05-features/02-universal-map/07-metrics-and-operators.md b/docs/zh/05-features/02-universal-map/07-metrics-and-operators.md index fcbe04f9..e2802ef9 100644 --- a/docs/zh/05-features/02-universal-map/07-metrics-and-operators.md +++ b/docs/zh/05-features/02-universal-map/07-metrics-and-operators.md @@ -69,7 +69,7 @@ permalink: /features/universal-map/metrics-and-operators - 客户端 ACK 缺失 - **现象**:服务端回复 SYN-ACK 后,客户端无响应,导致 TCP 建连失败 - **原因**: - - 客户端SYN Flood攻击 + - 客户端 SYN Flood 攻击 - 客户端端口扫描 - **建议**:确认是否为安全事件,并对异常客户端及时封堵。 - 客户端其他重置 @@ -103,7 +103,7 @@ permalink: /features/universal-map/metrics-and-operators - 检查防火墙策略 - 检查网络连通性 - 服务端其他重置 - - **现象**:服务端发送 SYN-ACK 后即发送 RST,导致 TCP 建连失败 + - **现象**:服务端发送 SYN-ACK 后即发送 RST,导致 TCP 建连失败 - **原因**:服务端操作系统异常 - **建议**:检查服务端操作系统日志 @@ -133,7 +133,7 @@ permalink: /features/universal-map/metrics-and-operators - **建议**: - 检查服务端应用状态 - 检查服务端操作系统日志 -- TCP连接超时 +- TCP 连接超时 - **现象**:传输过程中超过 300 秒无数据 - **原因**: - 客户端主机离线 @@ -147,11 +147,11 @@ permalink: /features/universal-map/metrics-and-operators #### TCP 断连异常 - 服务端半关 - - **现象**:服务端收到 FIN 之后,未回复FIN-ACK,TCP 四次挥手不完整 + - **现象**:服务端收到 FIN 之后,未回复 FIN-ACK,TCP 四次挥手不完整 - **原因**:服务端应用异常 - **建议**:检查服务端应用状态 - 客户端半关 - - **现象**:客户端收到 FIN 之后,未回复FIN-ACK,TCP 四次挥手不完整 + - **现象**:客户端收到 FIN 之后,未回复 FIN-ACK,TCP 四次挥手不完整 - **原因**:客户端应用异常 - **建议**:检查客户端应用状态 diff --git a/docs/zh/05-features/04-continuous-profiling/01-auto-profiling.md b/docs/zh/05-features/04-continuous-profiling/01-auto-profiling.md index 6cff0f90..c420eadb 100644 --- a/docs/zh/05-features/04-continuous-profiling/01-auto-profiling.md +++ b/docs/zh/05-features/04-continuous-profiling/01-auto-profiling.md @@ -40,6 +40,7 @@ permalink: /features/continuous-profiling/auto-profiling | rdma | C/C++ `*` | | ✔ | 说明: + - `*`: features in development - `**`: 运行 Java 程序的 JVM 须有符号表,参考[检查方法](#jvm-符号表检查) - `***`: 当前支持版本为 Python 3.10 @@ -57,6 +58,7 @@ permalink: /features/continuous-profiling/auto-profiling - 解释型语言:Python 获取 Profiling 数据需满足两个前提条件: + - 应用进程需要开启 Frame Pointer 或启用 Agent 的 DWARF 栈回溯能力 - 应用进程开启 Frame Pointer(帧指针寄存器): - 编译 C/C++:`gcc -fno-omit-frame-pointer` diff --git a/docs/zh/05-features/05-auto-tagging/04-custom-tags.md b/docs/zh/05-features/05-auto-tagging/04-custom-tags.md index fc5e9333..d60e5c0e 100644 --- a/docs/zh/05-features/05-auto-tagging/04-custom-tags.md +++ b/docs/zh/05-features/05-auto-tagging/04-custom-tags.md @@ -74,8 +74,8 @@ querier: # value for the custom tag. Here you can enter any tags seen in the results of # the `show tags from ` API. tag-fields: - - k8s.label.app - - auto_service + - k8s.label.app + - auto_service ``` 上述定义的 `$tag-name` 标签使用上与 `auto_instance`、`auto_service` 基本一致,额外的限制如下: diff --git a/docs/zh/05-features/05-auto-tagging/08-k8s-crd.md b/docs/zh/05-features/05-auto-tagging/08-k8s-crd.md index 3aed2b1e..2d7f0581 100644 --- a/docs/zh/05-features/05-auto-tagging/08-k8s-crd.md +++ b/docs/zh/05-features/05-auto-tagging/08-k8s-crd.md @@ -6,6 +6,7 @@ permalink: /features/auto-tagging/k8s-crd # 常见的特殊 K8s 资源或 CRD 当发现未同步的(找不到工作负载的)容器 Pod 时, + - 如果 Pod 的`metadata.ownerReferences[].apiVersion = apps.kruise.io/v1beta1`,那么对应的 K8s 平台应该是 OpenKruise - 如果 Pod 的`metadata.ownerReferences[].apiVersion = opengauss.sig/v1`,那么对应的 K8s 平台应该是 OpenGauss @@ -32,9 +33,9 @@ inputs: resources: kubernetes: api_resources: - - name: ingresses - disabled: true - - name: routes + - name: ingresses + disabled: true + - name: routes ``` 在 Agent 所在容器集群中修改 Agent 的 ClusterRole 配置,增加如下规则: @@ -76,12 +77,12 @@ inputs: resources: kubernetes: api_resources: - - name: clonesets - group: apps.kruise.io - - name: statefulsets - group: apps - - name: statefulsets - group: apps.kruise.io + - name: clonesets + group: apps.kruise.io + - name: statefulsets + group: apps + - name: statefulsets + group: apps.kruise.io ``` ::: tip @@ -115,7 +116,7 @@ inputs: resources: kubernetes: api_resources: - - name: opengaussclusters + - name: opengaussclusters ``` 在 Agent 所在容器集群中修改 Agent 的 ClusterRole 配置,增加如下规则: @@ -132,6 +133,7 @@ inputs: ``` # 其他 K8s 自定义资源 + ## 关于 Server 端 Lua 插件 由于一些用户的 Kubernetes 环境可能具有特殊的配置或安全要求,使得标准化的提取工作负载类型和工作负载名称的方式无法按预期工作,或者用户可能希望根据自己的逻辑来定制工作负载类型和工作负载名称。基此,DeepFlow 支持用户通过添加自定义的 lua 插件提取工作负载类型和工作负载名称。Lua 插件系统通过在固定的地方调用 Lua Function 获取一些用户自定义的工作负载类型和名称,提高 K8s 资源对接的灵活性与普适性。 diff --git a/docs/zh/06-guide/01-quick-start/01-5w-method.md b/docs/zh/06-guide/01-quick-start/01-5w-method.md index fec1140d..7bbc36f7 100644 --- a/docs/zh/06-guide/01-quick-start/01-5w-method.md +++ b/docs/zh/06-guide/01-quick-start/01-5w-method.md @@ -26,6 +26,7 @@ DeepFlow 可观测性平台通过 eBPF 采集以及开放的数据接口汇聚 m 什么是 “**5W 故障诊断方法**”? 在 DeepFlow 可观测性平台中,一般从宏观到微观有序调阅可观测性数据,逐步回答如下 5 个问题,可以快速、有效诊断出问题根因,称之为“**5W 故障诊断方法**”: + - **Who** is in trouble? - **When** is it in trouble? - **Which** request is in trouble? @@ -37,12 +38,14 @@ DeepFlow 可观测性平台通过 eBPF 采集以及开放的数据接口汇聚 m # Who & When? 在 DeepFlow 平台自动监测或人工分析应用服务相互调用的性能,发现应用服务的异常,回答 **Who** 和 **When** 的问题,即: + - 哪一个观测对象(比如 IT 系统中的某个容器服务、某个工作负载、某个 Pod、某一条路径等)的服务质量出现异常(出现响应错误、响应慢或超时)。 - 哪一个时间点/时间段出现了异常? ## 从哪里开始 通常从如下功能入口开始分析应用服务的性能: + - `追踪`-`资源分析`:应用服务(点)的性能指标分析入口,[指导链接](../ee-tenant/tracing/service-list/) - `追踪`-`路径分析`:应用服务访问路径(线)的性能指标分析入口,[指导链接](../ee-tenant/tracing/service-statistics/) - `追踪`-`拓扑分析`:应用服务访问拓扑(面)的分析入口,[指导链接](../ee-tenant/tracing/path-topology/) @@ -50,6 +53,7 @@ DeepFlow 可观测性平台通过 eBPF 采集以及开放的数据接口汇聚 m ## 如何开始搜索 通常使用命名空间(`pod_ns`)、容器服务(`pod_servce`)、工作负载(`pod_group`)、应用调用协议(`l7_protocol`)等不同维度的条件,并组合过滤,观测您所关注的对象的性能指标。 + - Step1:如果您负责某一个应用系统的运维,而应用模块部署并隔离在 K8s 的名字为 “A” 的命名空间(k8s namespace) 中,那么可以使用 `pod_ns = A` 来观测该业务系统所有应用服务的 RED 指标。 - Step2:如果你想在此基础之上仅观测 “A” 命名空间中名字为 “b” 的容器服务,则仅需再增加一个 `pod_svc = b` 的过滤条件。 - Step3:如果您想仅观测 http 协议调用的 RED 指标,则仅需增加一个 `l7_protocol = http` 的过滤条件。 @@ -57,6 +61,7 @@ DeepFlow 可观测性平台通过 eBPF 采集以及开放的数据接口汇聚 m ## 分析哪些性能指标 在 DeepFlow 平台使用 **RED** 指标(Rate、Error、Duration)作为评估业务质量/应用服务质量的核心指标: + - **Rate** (`请求速率`)——表征单位时间内接收的请求数量,用于衡量服务的吞吐量/压力。 - **Error** (`异常比例`)——表征所有请求中返回错误响应的比例,用于发现服务的异常,通常分为客户端原因导致的 Error 和 服务端原因导致的 Error,而服务端原因导致的 Error 通常是应用的关注重点。 - **Duration** (`响应时延`)——表征从请求到响应消耗的时间,用于发现服务响应慢的情况;通常使用`响应时延均值`、`响应时延 P95`、`响应时延 P99`等观测响应时延的统计结果。 @@ -84,6 +89,7 @@ DeepFlow 平台为每一个观测对象提供了隐藏的`右滑窗`,在**指 ## 什么是调用链追踪 基于 eBPF 技术 DeepFlow 创新实现了零侵扰的分布式追踪,即无需生成、无需注入、无需传播 TraceID 即可实现分布式追踪: + - [功能使用指导链接](../ee-tenant/tracing/call-chain-tracing/) - [B 站学习视频——3 分钟理解 DeepFlow 调用链追踪火焰图](https://www.bilibili.com/video/BV1di421k7JE/) - [B 站学习视频——3 分钟理解 DeepFlow 调用链追踪实现原理](https://www.bilibili.com/video/BV1ZC411E7ad/) @@ -171,6 +177,7 @@ DeepFlow 平台为每一个观测对象提供了隐藏的`右滑窗`,在**指 # What? 通过 DeepFlow 平台的应用调用链追踪回答了“**Where is the root position?**” 的问题之后,下一步便是围绕 **Root Position** 展开多维度的数据分析,继续回答 “**What**” 的问题(**What is the root cause?**): + - 当通过调用链追踪做定界定位,确定问题边界是某一个应用进程后,即可进入**应用诊断**环节,对应用实例进行多个维度数据的分析诊断,确定应用故障的根因。 - 当通过调用链追踪做定界定位,确定问题边界是网络传输的原因后,即可进入**网络诊断**环节,对网络传输进行多个维度数据的分析诊断,确定网络故障的根因。 - 当确定问题与系统性能相关时(比如系统 CPU 用量、系统 Load、系统网络接口),还可进入**系统诊断**环节,对操作系统进行多个维度数据的分析诊断,确定操作系统故障的根因。 @@ -266,6 +273,7 @@ DeepFlow 平台为每一个观测对象提供了隐藏的`右滑窗`,在**指 ### 网络指标分析 当调用链追踪中确定某个`网络 Span` 是 Root Postion 后,随即一键调阅该次应用调用关联的`网络性能`, 进而确定 TCP 会话在三次握手、TLS 建连、数据交互、系统响应等不同过程的时延,确定网络传输慢的关键原因: + - `TCP 建连时延`——TCP 三次握手过程的时延; - `TLS 建连时延`——TLS 建连过程的时延; - `平均数据时延`——请求 Data 到响应 Data 的时延(多次过程的平均值) diff --git a/docs/zh/08-integration/01-process/01-wasm-plugin.md b/docs/zh/08-integration/01-process/01-wasm-plugin.md index 59949cfe..1d9980bf 100644 --- a/docs/zh/08-integration/01-process/01-wasm-plugin.md +++ b/docs/zh/08-integration/01-process/01-wasm-plugin.md @@ -21,6 +21,7 @@ Wasm 插件系统通过在固定的地方调用 Wasi Export Function 实现一 关于 Wasm Plugin 的开发你也可以参考这篇博客文章:[使用 DeepFlow Wasm 插件实现业务可观测性](https://www.deepflow.io/blog/zh/035-deepflow-enabling-zero-code-observability-for-applications-by-webAssembly/)。 对于 HTTP2 和 gRPC 协议,deepflow-agent 已经内置了完整的 Header 字段解析能力,并且你可以通过 agent-group-config 来让 deepflow-agent 采集指定的 Header 字段。因此 HTTP2/gRPC 的 Wasm Plugin 只需要对 Payload 进行解析即可。需要注意的是,由于存在 cBPF/eBPF-kprobe(压缩 Header + 原始 Payload)和 eBPF-uprobe(原始 Header + 原始 Payload)两种采集方式,插件的编写方法有些差别: + - 对于 cBPF/eBPF-kprobe 采集的数据,直接参照上表中的 `http` 插件来编写,解析 Payload。 - 对于 eBPF-uprobe 采集的数据,目前仅支持视为新的协议在 Plugin 中重新解析,参考上表中的 `go_http2_uprobe` 插件来编写(对增强的支持还在开发中)。 diff --git a/docs/zh/08-integration/02-input/01-metrics/02-prometheus.md b/docs/zh/08-integration/02-input/01-metrics/02-prometheus.md index 7fc0cc83..93c06c2e 100644 --- a/docs/zh/08-integration/02-input/01-metrics/02-prometheus.md +++ b/docs/zh/08-integration/02-input/01-metrics/02-prometheus.md @@ -105,7 +105,7 @@ inputs: x_exporter: type: prometheus_scrape endpoints: - - http://${HOST:PORT}/metrics + - http://${HOST:PORT}/metrics scrape_interval_secs: 10 scrape_timeout_secs: 10 honor_labels: true @@ -115,7 +115,7 @@ inputs: prometheus_remote_write: type: prometheus_remote_write inputs: - - x_exporter + - x_exporter endpoint: http://127.0.0.1:38086/api/v1/prometheus healthcheck: enabled: false diff --git a/docs/zh/08-integration/02-input/02-tracing/03-apm-trace-api.md b/docs/zh/08-integration/02-input/02-tracing/03-apm-trace-api.md index df7a2ca8..4d369cd1 100644 --- a/docs/zh/08-integration/02-input/02-tracing/03-apm-trace-api.md +++ b/docs/zh/08-integration/02-input/02-tracing/03-apm-trace-api.md @@ -58,24 +58,24 @@ app: 其中,从 SkyWalking APM 的 [skywalking-query-protocol](https://github.com/apache/skywalking-query-protocol/blob/master/trace.graphqls)转换为 DeepFlow 火焰图的映射关系如下表: -| 名称 | 中文 | SkyWalking 数据结构 | 描述 | -|-----------------|--------------|-----------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------------------| -| startTimeUs | 开始时间 | span.startTime | -- | -| endTimeUs | 结束时间 | span.endTime | -- | -| tapSide | 观测点 | span.spanType.Exit: 客户端应用(C-APP), span.spanType.Entry: 服务端应用(S-APP), span.spanType.Local: 应用(APP) | 观测点,即 observation_point,转换为对应的枚举值 | -| traceID | TraceID | span.traceID | -- | -| spanID | SpanID | span.segmentID-span.spanID | -- | -| parentSpanID | ParentSpanID | span.segmentID-span.parentSpanID/span.ref.parentSegmentID/span.ref.parentSpanID | 当 span.parentSpanID=-1时,尝试获取 span.ref 作为 parentSpan | -| spanKind | span 类型 | span.type=Exit: SPAN_KIND_CLIENT, span.type=Entry: SPAN_KIND_SERVER, span.type=Local: SPAN_KIND_INTERNAL | 转换为对应的枚举值 | -| endpoint | 请求端点 | span.endpointName | 请求的具体资源,在 HTTP 协议中一般是请求路由 | -| appService | 应用服务 | span.serviceCode | -- | -| appInstance | 应用服务实例 | span.serviceInstance | -- | -| serviceUname | 服务名称 | span.serviceCode | -- | -| requestType | 请求类型 | span.tags.[http.method/cache.cmd/db.operation/rpc.method] | 根据协议获取 tag 中的 value | -| requestResource | 请求资源 | span.endpointName/span.tags.[db.statement/cache.key/url/http.url] | 当 span.endpointName 不存在时,尝试从 http.url 中截取,仅提取域名之后的请求信息 | -| responseCode | 响应码 | span.tags.[http.status.code/http.status_code/http.status] | -- | -| reponseStatus | 响应状态 | -- | 从 responseCode 转换,2xx~3xx: STATUS_OK, 4xx: STATUS_CLIENT_ERROR, 5xx: STATUS_SERVER_ERROR | -| signalSource | 信号源 | OTEL(4) | 固定枚举值 | -| l7Protocol | 应用协议 | span.layer=Http: HTTP, span.tags.[db.type/db.system/http.scheme/rpc.system/messaging.system/messaging.protocol] | 若 span.layer 不等于 Http,则尝试从 span.tags 中获取,如果有任意 `http.` 开头的 tag,均视为 HTTP 协议 | -| l7ProtocolStr | 应用协议(名称) | -- | 根据 l7Protocol 的枚举值获取具体名称 | -| attribute | 标签 | span.tags | -- | +| 名称 | 中文 | SkyWalking 数据结构 | 描述 | +| --------------- | ---------------- | --------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- | +| startTimeUs | 开始时间 | span.startTime | -- | +| endTimeUs | 结束时间 | span.endTime | -- | +| tapSide | 观测点 | span.spanType.Exit: 客户端应用(C-APP), span.spanType.Entry: 服务端应用(S-APP), span.spanType.Local: 应用(APP) | 观测点,即 observation_point,转换为对应的枚举值 | +| traceID | TraceID | span.traceID | -- | +| spanID | SpanID | span.segmentID-span.spanID | -- | +| parentSpanID | ParentSpanID | span.segmentID-span.parentSpanID/span.ref.parentSegmentID/span.ref.parentSpanID | 当 span.parentSpanID=-1 时,尝试获取 span.ref 作为 parentSpan | +| spanKind | span 类型 | span.type=Exit: SPAN_KIND_CLIENT, span.type=Entry: SPAN_KIND_SERVER, span.type=Local: SPAN_KIND_INTERNAL | 转换为对应的枚举值 | +| endpoint | 请求端点 | span.endpointName | 请求的具体资源,在 HTTP 协议中一般是请求路由 | +| appService | 应用服务 | span.serviceCode | -- | +| appInstance | 应用服务实例 | span.serviceInstance | -- | +| serviceUname | 服务名称 | span.serviceCode | -- | +| requestType | 请求类型 | span.tags.[http.method/cache.cmd/db.operation/rpc.method] | 根据协议获取 tag 中的 value | +| requestResource | 请求资源 | span.endpointName/span.tags.[db.statement/cache.key/url/http.url] | 当 span.endpointName 不存在时,尝试从 http.url 中截取,仅提取域名之后的请求信息 | +| responseCode | 响应码 | span.tags.[http.status.code/http.status_code/http.status] | -- | +| reponseStatus | 响应状态 | -- | 从 responseCode 转换,2xx~3xx: STATUS_OK, 4xx: STATUS_CLIENT_ERROR, 5xx: STATUS_SERVER_ERROR | +| signalSource | 信号源 | OTEL(4) | 固定枚举值 | +| l7Protocol | 应用协议 | span.layer=Http: HTTP, span.tags.[db.type/db.system/http.scheme/rpc.system/messaging.system/messaging.protocol] | 若 span.layer 不等于 Http,则尝试从 span.tags 中获取,如果有任意 `http.` 开头的 tag,均视为 HTTP 协议 | +| l7ProtocolStr | 应用协议(名称) | -- | 根据 l7Protocol 的枚举值获取具体名称 | +| attribute | 标签 | span.tags | -- | diff --git a/docs/zh/08-integration/02-input/04-log/02-vector.md b/docs/zh/08-integration/02-input/04-log/02-vector.md index 11d65752..d8fe05bd 100644 --- a/docs/zh/08-integration/02-input/04-log/02-vector.md +++ b/docs/zh/08-integration/02-input/04-log/02-vector.md @@ -202,13 +202,13 @@ transforms: tag_log: # inputs 指的是数据来源,这里可配置 sources 中的 Key,也可以配置 transforms 中的其他 Key inputs: - - nginx_logs + - nginx_logs # ... flush_log: # tag_log 来源于上一个 transforms 模块,这样一份数据就按顺序被两个 transforms 模块处理 inputs: - - tag_log - - file_logs + - tag_log + - file_logs # ... # 数据输出 @@ -216,8 +216,8 @@ sinks: push_log: # 同样,这里的 inputs 可以同时来自 sources 模块或 transforms 模块 inputs: - - flush_log - - kubernetes_logs + - flush_log + - kubernetes_logs # ... ``` diff --git a/docs/zh/08-integration/03-output/01-query/04-mcp-server.md b/docs/zh/08-integration/03-output/01-query/04-mcp-server.md index 13a2b443..eac7aa31 100644 --- a/docs/zh/08-integration/03-output/01-query/04-mcp-server.md +++ b/docs/zh/08-integration/03-output/01-query/04-mcp-server.md @@ -32,13 +32,13 @@ DeepFlow 部署(参考官网[部署文档](https://deepflow.io/docs/zh/ce-inst 为了实现基于 git commit id 的性能数据查询,需要在应用 Pod 上注入 git_commit_id label。生产环境建议通过 CI/CD 流程自动注入,本教程以手动修改 YAML 为例: ```yaml - template: - metadata: - labels: - git_commit_id: 7ea306a6dca26d54e65e350439cf8bd0d41c9482 +template: + metadata: + labels: + git_commit_id: 7ea306a6dca26d54e65e350439cf8bd0d41c9482 ``` -**Step 3 Cursor 中配置 DeepFlow MCP Server** +**Step 3 Cursor 中配置 DeepFlow MCP Server** 在项目目录下的 .cursor 文件夹下,增加 mcp.json 文件,输入如下内容(其中 mcp server 端口默认为 20080): @@ -46,7 +46,7 @@ DeepFlow 部署(参考官网[部署文档](https://deepflow.io/docs/zh/ce-inst { "mcpServers": { "DeepFlow_Git_Commit_Profile": { - "url": "http://$deepflow_controller_ip:20080/mcp", + "url": "http://$deepflow_controller_ip:20080/mcp", "headers": {} } } @@ -57,4 +57,4 @@ DeepFlow 部署(参考官网[部署文档](https://deepflow.io/docs/zh/ce-inst 在 Cursor AI Chat 中输入需要分析的 commit id,AI 将自动获取该版本的性能分析报告。注意:确保该 commit 对应的应用已部署到 DeepFlow 监控环境中,且已采集到性能剖析数据。 -![Cursor AI Chat](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20250626115236.png) \ No newline at end of file +![Cursor AI Chat](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20250626115236.png) diff --git a/docs/zh/10-release-notes/08-ee-6.6-release.md b/docs/zh/10-release-notes/08-ee-6.6-release.md index 40862fce..ecf78052 100644 --- a/docs/zh/10-release-notes/08-ee-6.6-release.md +++ b/docs/zh/10-release-notes/08-ee-6.6-release.md @@ -64,7 +64,7 @@ permalink: /release-notes/release-6.6-ee - ⭐ 调用链追踪支持`瀑布列表`的显示模式。 - ⭐ deepflow-agent 支持直接接收 SkyWalking、Datadog 的追踪数据,无需经过 otel-collector 转发。 - ⭐ 调用链追踪火焰图自动校正不同机器的细微时钟偏差。 - - ⭐优化`网络路径`右滑页的易用性。 + - ⭐ 优化`网络路径`右滑页的易用性。 - 优化开启 eBPF uprobe 功能的进程配置能力,[文档](../configuration/agent/#inputs.proc.process_matcher)。 - 优化调用链追踪页面:左侧快速过滤框顺序调整、增加趋势分析折线图、优化调用日志详情表格。 - 资源分析、路径分析、拓扑分析页面支持自动选择最合适的指标时间粒度进行查询,以优化查询速度。 @@ -117,7 +117,7 @@ permalink: /release-notes/release-6.6-ee - 性能提升 - 优化 NAT 追踪 API 性能。 - 易用性提升 - - ⭐优化`网络路径`右滑页的易用性。 + - ⭐ 优化`网络路径`右滑页的易用性。 - 页面上所有流量速率的默认单位从字节每秒(`Bps`)修改为比特每秒(`bps`)。 - 网络流日志(`l4_flow_log`)中的非 TCP 流量,将其结束状态(`close_type`)从超时调整为正常结束(1)。 - 资源分析、路径分析、拓扑分析页面支持自动选择最合适的指标时间粒度进行查询,以优化查询速度。 @@ -252,6 +252,7 @@ N/A - 性能提升 - ClickHouse 通过代理访问 MySQL 获取字典数据,降低 MySQL 连接数、优化跨 Region 带宽消耗。 - Agent + - ⭐ OneAgent:支持使用 deepflow-agent 采集应用日志、主机系统指标、K8s 容器系统指标。 - ⭐ OneAgent:支持使用 deepflow-agent 进行持续拨测。 - ⭐ 安全性:支持限制 deepflow-agent 使用的 Socket 数量,[文档](../configuration/agent/#global.limits.max_sockets)。 @@ -276,6 +277,7 @@ N/A - 合并用于 agent 自监控的、传输 deepflow_stats 和 agent_log 数据使用的 Socket。 - 合并集成 Prometheus 和 Telegraf 时,传输 prometheus 和 telegraf 指标使用的 Socket。 - 性能提升 + - ⭐ 降低 Agent 的 eBPF 内核内存开销,默认配置下可降低 60% 的内存消耗。 - ⭐ 支持利用 BPF FANOUT 机制提升采集性能,[文档](../configuration/agent/#inputs.cbpf.af_packet.tunning.packet_fanout_count)。 - ⭐ 性能:优化 Agent 中用于应用性能指标的 Cache 的内存占用,通过及时清理失效的 LRU 条目,测试环境下可见整体内存消耗降低 **43%**。 @@ -286,8 +288,10 @@ N/A - 支持完全关闭 cBPF 数据采集(通过配置 `inputs.cbpf.af_packet.interface_regex` 为空字符串),以降低内存开销,[文档](../configuration/agent/#inputs.cbpf.af_packet.interface_regex)。 - 支持 deepflow-agent 使用一个 Socket 传输所有观测数据,[文档](../configuration/agent/#outputs.socket.multiple_sockets_to_ingester)。 - 在 Linux 启用了 BTF(BPF Type Format)的情况下,当 X86 架构下内核大于等于 [5.5](https://github.com/torvalds/linux/commit/f1b9509c2fb0ef4db8d22dac9aef8e856a5d81f6)、ARM 架构下内核大于等于 [6.0](https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git/commit/?h=linux-6.0.y&id=efc9909fdce00a827a37609628223cd45bf95d0b) 时,agent 将会自动使用 fentry/fexit 替代 kprobe/kretprobe,此时可获得约 15% 的性能提升。 + - 支持 Watchdog 机制,在极端情况下保证熔断能够正常执行。 - 支持压缩发送应用日志,压缩比例在 5:1~20:1 之间,[文档](../configuration/agent/#outputs.compression.application_log)。 + - 易用性改进 - 大幅度缩减 Web 页面的 URL 长度,在 URL 中记忆 Web 页面右滑框的激活状态。 - 采集器列表中展示所属容器集群。 diff --git a/translate/translated/01-about/06-users.md b/translate/translated/01-about/06-users.md index b5ffb743..30391b3f 100644 --- a/translate/translated/01-about/06-users.md +++ b/translate/translated/01-about/06-users.md @@ -5,39 +5,59 @@ permalink: /about/users > This document was translated by ChatGPT +# 2025 + +| Industry | User | Source | Title | Links | +| -------- | ---------- | ------ | ------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Cloud Service | Tencent Cloud | Meetup | Observability Practice of DeepFlow in Tencent TKE Internal Platform | [Article](https://mp.weixin.qq.com/s/tsVObqnUBOQ-fE6uK6_oxA), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/85bfdb75cf8v77d8b618bf2h90a769b4_20241217152025.pdf), [Video](https://www.bilibili.com/video/BV1y2kEYkEf1) | +| New Manufacturing | An Intelligent Car Company | Article | Slow Call Investigation Record: Efficient Triage of Service Mesh Sidecar Performance Bottleneck | [Article](https://mp.weixin.qq.com/s/0QtKqQDuV1KjYYCPAP4sBQ) | +| Internet | Kingsoft Office | Meetup | Observability Practice of Kingsoft Office Based on DeepFlow Docker Mode | [Article](https://mp.weixin.qq.com/s/Pd1-lO9pjAKhofzmpuEBgw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/jpg/6815b7d64b37417156ca9f8bf1403705_20250702104904.pdf), [Video](https://www.bilibili.com/video/BV1AogSz9E3o) | + # 2024 -| Industry | User | Source | Title | Link | -| -------- | --------- | ------ | ------------------------------------------------------------ | --------------------------------------------------------- | -| Finance | FinPoints | Article | FinPoints x DeepFlow: How to Achieve SRE 99.9% Service Level Objective (SLO) | [Article](https://mp.weixin.qq.com/s/WoGDcmT1ua3N3DXa11Bk4g) | -| Internet | Qimai Technology | Article | Qimai Technology x DeepFlow: Observability Platform Practices Behind Explosive Business Growth | [Article](https://mp.weixin.qq.com/s/P2tMeAYCMns05zG8nfj6dg) | -| Consumer Electronics | A Mobile Phone Manufacturer | Article | Using DeepFlow to Eliminate "Going in the Wrong Direction" in APISIX Fault Diagnosis | [Article](https://mp.weixin.qq.com/s/a-x_ce6VO-L1SaXs8PKoAg) | -| Cloud Services | Tencent Cloud | Article | Observability Practices of a Tencent Cloud Business Based on DeepFlow | [Article](https://mp.weixin.qq.com/s/57e3dAvN9gYcwWGjt-BMbw) | -| Gaming | Tencent Interactive Entertainment | Article | Advanced Zero-Interference Observability Practices Based on DeepFlow in Tencent Games | [Article](https://mp.weixin.qq.com/s/6v5jPLSMD1SZJITIKvHpWA) | -| Telecom | China Mobile | Article | Out-of-the-Box eBPF Observability: China Mobile Pangea PaaS Platform Case Study | [Article](https://mp.weixin.qq.com/s/Byb_PJ7hlUAeTotAamgqRA) | -| Banking | A State-Owned Bank | Article | Achieving Full-Chain Observability of Distributed Database TDSQL with Zero Interference Using DeepFlow | [Article](https://mp.weixin.qq.com/s/IJntZDqBpLOWP2-JGY6Hmw) | -| Telecom | China Mobile | Article | DeepFlow Metadata Database PostgreSQL Transformation Practices | [Article](https://mp.weixin.qq.com/s/1_8939kNHZjqrABB9nlzBg) | -| Banking | China Minsheng Bank | Meetup | eBPF Observability Practices for Cloud-Native Business in China Minsheng Bank | [Article](https://mp.weixin.qq.com/s/9XctB-EPqOPSbK1YL2JzlQ), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ebae4e2d4d0ea71c28228c5e0dbb8f23_20231225162831.pdf), [Video](https://www.bilibili.com/video/BV1ag4y1C7DD) | -| Gaming | Tencent Interactive Entertainment | Meetup | Eliminate Blind Spots! True Full-Stack Observability Practices in Tencent Games | [Article](https://mp.weixin.qq.com/s/vzRebv7TMrrRi8TUV9qj5A), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/580f8117457f0e2bbc2f3818f7d42300_20231225162841.pdf), [Video](https://www.bilibili.com/video/BV1ku4y1K7PF) | +| Industry | User | Source | Title | Links | +| -------- | ---------- | ------ | ---------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Finance | A Leading Brokerage | Article | Observability in Practice: From Clarity to Rapidly Pinpointing High Business Response Latency Issues | [Article](https://mp.weixin.qq.com/s/4ZRfRlgHw2DWaSuFaHiJDg) | +| Banking | A Joint-Stock Bank | Article | eBPF Zero-Intrusion Distributed Tracing: Locking Java Program I/O Thread Blockage in 3 Minutes | [Article](https://mp.weixin.qq.com/s/8998CkTGrvPwoad5wkqitA) | +| Banking | A Joint-Stock Bank | Article | Fault Diagnosis: Locking Distributed Core Database in 3 Minutes to Accelerate Fintech Development, Testing, and Migration | [Article](https://mp.weixin.qq.com/s/VfoPeKp-iMeQJc2VMEAZtA) | +| Banking | A State-Owned Bank | Article | Diagnosing Probabilistic Transaction Failures Caused by Tomcat TCP Timeout Configuration Errors in 3 Minutes | [Article](https://mp.weixin.qq.com/s/lao6SRU6xwo0ImEAqlNbfQ) | +| Banking | A Joint-Stock Bank | Article | DeepFlow Large Model Agent: Locating Java Program Hang Faults in 3 Minutes | [Article](https://mp.weixin.qq.com/s/1H3mqKRL0GBrE3qstfqKlQ) | +| Telecom | China Mobile | Article | In-Depth Analysis of How DeepFlow Collects Business Metrics of Large Model Services | [Article](https://mp.weixin.qq.com/s/GjIKMIaDxbxNo75uhvTgAg) | +| Gaming | Tencent Interactive Entertainment | Meetup | Blue Whale Observability Platform: Exploring Unified Observability Data Association Models | [Article](https://mp.weixin.qq.com/s/-osVmY1V6yAycVx6d3CAMA), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/721e3ac62b51e6234eb10f03e7d41629_20240914102155.pdf), [Video](https://www.bilibili.com/video/BV1ZJ46eiE3S) | +| Internet | Kingsoft Office | Meetup | Zero-Intrusion Observability Practice of Kingsoft Office Based on DeepFlow | [Article](https://mp.weixin.qq.com/s/7M5BCzDDQ3NQmuieIeCqXw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/3838d594c942dc4765a223573206e5b5_20240913152317.pdf), [Video](https://www.bilibili.com/video/BV1JV46e6ErU) | +| Finance | Futu Securities | Article | From Deployment to Optimization: Futu Securities' Exploration Journey with DeepFlow | [Article](https://mp.weixin.qq.com/s/xFBiyTRrADUnCOMPwrTOQw) | +| Finance | FinPoints | Article | FinPoints x DeepFlow: Achieving SRE 99.9% Service Level Objective (SLO) | [Article](https://mp.weixin.qq.com/s/WoGDcmT1ua3N3DXa11Bk4g) | +| Internet | Qimai Technology | Article | Qimai Technology x DeepFlow: Observability Platform Practice Behind Explosive Business Growth | [Article](https://mp.weixin.qq.com/s/P2tMeAYCMns05zG8nfj6dg) | +| Consumer Electronics | A Smartphone Manufacturer | Article | Using DeepFlow to Eliminate "Going in the Wrong Direction" in APISIX Fault Diagnosis | [Article](https://mp.weixin.qq.com/s/a-x_ce6VO-L1SaXs8PKoAg) | +| Cloud Service | Tencent Cloud | Meetup | Observability Practice of a Tencent Cloud Business Based on DeepFlow | [Article](https://mp.weixin.qq.com/s/57e3dAvN9gYcwWGjt-BMbw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/52a0ea94c84600ddc34c53e10e048420_20240802114858.pdf), [Video](https://www.bilibili.com/video/BV1q4421Z7ni) | +| Gaming | Tencent Interactive Entertainment | Article | Advanced Zero-Intrusion Observability Practice of Tencent Games Based on DeepFlow | [Article](https://mp.weixin.qq.com/s/6v5jPLSMD1SZJITIKvHpWA) | +| Telecom | China Mobile | Article | Out-of-the-Box eBPF Observability: China Mobile PaaS Platform Case Study | [Article](https://mp.weixin.qq.com/s/Byb_PJ7hlUAeTotAamgqRA) | +| Banking | A State-Owned Bank | Article | Zero-Intrusion Full-Stack Observability of Distributed Database TDSQL with DeepFlow | [Article](https://mp.weixin.qq.com/s/IJntZDqBpLOWP2-JGY6Hmw) | +| Telecom | China Mobile | Meetup | DeepFlow Metadata Database PostgreSQL Transformation Practice | [Article](https://mp.weixin.qq.com/s/1_8939kNHZjqrABB9nlzBg), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/713b09f77232c733ff17d2c81955d5f6_20240802124358.pdf), [Video](https://www.bilibili.com/video/BV1q4421Z7ni) | +| Banking | China Minsheng Bank | Meetup | eBPF Observability Practice for Cloud-Native Business at China Minsheng Bank | [Article](https://mp.weixin.qq.com/s/rcCSDZfauhDdRD32hf5oxw), [PPT](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ebae4e2d942ea71c56a28c5b0dbb8f23_20240913152317.pdf), [Video](https://www.bilibili.com/video/BV1ag4y1C7DD) | +| Gaming | Tencent Interactive Entertainment | Meetup | Eliminating Blind Spots: Advanced Observability Practice on Blue Whale Platform | [Article](https://mp.weixin.qq.com/s/6v5jPLSMD1SZJITIKvHpWA), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/580e3ac62b51e6234eb10f03e7d41629_20240914102155.pdf), [Video](https://www.bilibili.com/video/BV1ku4y1K7PF) | # 2023 -| Industry | User | Source | Title | Link | -| -------- | -------- | ------ | ------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Telecom | China Mobile | Meetup | eBPF-Based Application Observability Practices on China Mobile Pangea PaaS Platform | [Article](https://mp.weixin.qq.com/s/ACS4AXFUk0uCXAsVTBi2SQ) | -| Community | Zheng Zhicong | Meetup | DeepFlow Protocol Parsing Extension Practices | [Article](https://mp.weixin.qq.com/s/GvUwamT-1VYHZQW34JBdow), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/50259d1f763207ff241a31b17231b871_20231201173751.pdf), [Video](https://www.bilibili.com/video/BV1pc411q7WH) | -| E-commerce | Zcygov | Meetup | Observability Practices in Zcygov | [Article](https://mp.weixin.qq.com/s/P_r1LQ3HerYNBYPZPClc2g), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/7698944121a1ce331c35428be49c2975_20230921103323.pdf), [Video](https://www.bilibili.com/video/BV1Sw411e7zC) | -| E-commerce | Weipaitang | Meetup | Building a Zero-Interference Observability Platform Based on DeepFlow in Weipaitang | [Article](https://mp.weixin.qq.com/s/P1tsmFW_9poIScxXCdOlLg), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ab5c0568c000db0d0669c8c6a59c3551_20230921103335.pdf), [Video](https://www.bilibili.com/video/BV1zH4y1S7zG) | -| Banking | China Everbright Bank | Article | A Brief Discussion on Performance Tuning of Distributed Systems - Overlay Layer Packet Analysis | [Article](https://mp.weixin.qq.com/s/aXwH6IIjCwZYHHqtqP2NSQ) | -| Banking | China Minsheng Bank | Article | Cloud-Native Observability Solutions Assist China Minsheng Bank's IT System Security Operations | [Article](https://mp.weixin.qq.com/s/rcCSDZfauhDdRD32hf5oxw) | -| Gaming | Quwan Technology | Article | Complete Guide: How to Compile, Package, and Deploy a Secondary Development of DeepFlow | [Article](https://mp.weixin.qq.com/s/-jWYq2rTRaTueuN0sAb3lA) | -| Community | Luga Lee | Article | Understanding the Automated Observability Platform Based on eBPF - DeepFlow | [Article](https://mp.weixin.qq.com/s/vkHsvoxJ6Ep-githtJAv7g) | -| Gaming | Tencent Interactive Entertainment | Meetup | Exploration and Practices of DeepFlow in Tencent Blue Whale Observability Platform | [Article](https://www.infoq.cn/article/raua40qhu5ejhmqb0mf3), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/1de79730a61f2f03dce9890862733cf4_20231031154518.pdf), [Video](https://www.bilibili.com/video/BV1o14y1S7iy) | -| Internet | Xiaomi Group | Meetup | Current Status and Challenges of DeepFlow Implementation in Xiaomi | [Article](https://mp.weixin.qq.com/s/0WMIdy1SoTYRTkU2e-PprQ), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a1ee4bcf5678dbd276353f4b59f4aeff_20231031154555.pdf), [Video](https://www.bilibili.com/video/BV12u411h7bn) | -| Software | Alauda | Meetup | A Three-Month K8s DNS Troubleshooting Process | [Article](https://mp.weixin.qq.com/s/dDfckiTaALmFYHL6Tes_SA), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ff69a942735788d654ba3b7d5acc24c6_20231031154454.pdf), [Video](https://www.bilibili.com/video/BV13X4y147UN) | +| Industry | User | Source | Title | Links | +| -------- | ---------- | ------ | ------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| New Manufacturing | An Intelligent Car Company | Live Stream | [Observability in Practice] Quickly Locating K8s CNI Port Issues | [Article](https://mp.weixin.qq.com/s/Ex7o_n4dh0VgkPFYGCVFQ), [Video](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a1ee4e2d4d37417156ca9f8bf1403705_20230815170039.pdf) | +| New Manufacturing | An Intelligent Car Company | Live Stream | [Observability in Practice] Quickly Locating K8s CNI Port Issues | [Article](https://mp.weixin.qq.com/s/fzjbR8rlIOLd1eH0XDvM_w), [Video](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a1ee4e2d4b37417156ca9f8bf1403705_20230815170039.pdf) | +| Logistics | A Logistics Company | Live Stream | [Observability in Practice] Quickly Locating K8s CNI Port Issues | [Article](https://mp.weixin.qq.com/s/fzjbR8rlIOLd1eH0XDvM_w), [Video](https://www.bilibili.com/video/BV1VX4y177pG), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a1ee4e2d4d0ea71c56ca9f8c6a59c3551_20230815170039.pdf) | +| Telecom | China Mobile | Meetup | Application Observability Practice on China Mobile PaaS Platform Based on eBPF | [Article](https://mp.weixin.qq.com/s/ACS4AXFUk0uCXAsVTBi2SQ) | +| Community | Zheng Lee | Meetup | Understanding Automated Observability Platform Based on eBPF - DeepFlow | [Article](https://mp.weixin.qq.com/s/vkHsvoxJ6Ep-githtJAv7g) | +| E-commerce | ZC Cloud | Meetup | ZC Cloud's Observability Practice Behind Explosive Business Growth with DeepFlow | [Article](https://mp.weixin.qq.com/s/P2tMeAYCMns05zG8nfj6dg), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/7698944121a1ce331c354f8be49c2975_20230921152317.pdf), [Video](https://www.bilibili.com/video/BV1Sw411e7zC) | +| E-commerce | Weipaitang | Meetup | Weipaitang's Zero-Intrusion Observability Practice with DeepFlow | [Article](https://mp.weixin.qq.com/s/7GVplyh_pspcJ7c2qP2NSQ), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ff69a942735788d1c65428b7d5acc24c6_20230921103335.pdf), [Video](https://www.bilibili.com/video/BV1zH4y1S7zG) | +| Banking | Everbright Bank | Article | Cloud-Native Observability Pathway of Everbright Bank with eBPF | [Article](https://mp.weixin.qq.com/s/aXwH6IIjCwZYHHqtqP2NSQ) | +| Banking | Minsheng Bank | Article | Cloud-Native Observability Practice Assisting IT System Security Operations at Minsheng Bank | [Article](https://mp.weixin.qq.com/s/rcCSDZfauhDdRD32hf5oxw) | +| Gaming | Quwan Technology | Article | Quwan Technology's Journey with DeepFlow: From Deployment to Optimization | [Article](https://mp.weixin.qq.com/s/-jWYq2rTRaTueuN0sAb3lA) | +| Community | Luga Lee | Article | Understanding the Automated Observability Platform Based on DeepFlow | [Article](https://mp.weixin.qq.com/s/vkHsvoxJ6Ep-githtJAv7g) | +| Gaming | Tencent Interactive Entertainment | Meetup | DeepFlow in Tencent TKE Internal Platform Observability Practice | [Article](https://mp.weixin.qq.com/s/tsVObqnUBOQ-fE6uK6_oxA), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/85bfdb75cf8v77d8b618bf2h90a769b4_20241217152025.pdf), [Video](https://www.bilibili.com/video/BV1y2kEYkEf1) | +| Internet | Xiaomi Group | Meetup | DeepFlow Observability Practice in Xiaomi Group | [Article](https://mp.weixin.qq.com/s/7M5BCzDDQ3NQmuieIeCqXw), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/a1ee4bcf5678dbd276353f4b59f4aeff_20231031154555.pdf), [Video](https://www.bilibili.com/video/BV12u411h7bn) | +| Software | Alauda | Meetup | A Three-Month Journey of Troubleshooting K8s DNS Issues | [Article](https://mp.weixin.qq.com/s/dDfckiTaALmFYHL6Tes_SA), [PPT](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/pdf/ff69a942735788d654ba3b7d5acc24c6_20231031154454.pdf), [Video](https://www.bilibili.com/video/BV13X4y147UN) | # 2022 -| Industry | User | Source | Title | Link | -| -------- | -------- | ------ | ------------------------------------- | --------------------------------------------------------- | -| Banking | China Everbright Bank | Article | Thoughts on the eBPF Cloud Observability Path of China Everbright Bank | [Article](https://mp.weixin.qq.com/s/7GVplyh_pspcJ7c9qmfyOg) | \ No newline at end of file +| Industry | User | Source | Title | Links | +| -------- | --------- | ------ | ------------------------------------ | ---------------------------------------------------------- | +| Banking | Everbright Bank | Article | Reflections on the Path of eBPF Cloud Observability at Everbright Bank | [Article](https://mp.weixin.qq.com/s/7GVplyh_pspcJ7c9qmfyOg) | \ No newline at end of file diff --git a/translate/translated/02-ce-install/01-overview.md b/translate/translated/02-ce-install/01-overview.md index 00c9ebb2..782077a1 100644 --- a/translate/translated/02-ce-install/01-overview.md +++ b/translate/translated/02-ce-install/01-overview.md @@ -5,13 +5,13 @@ permalink: /ce-install/overview > This document was translated by ChatGPT -This chapter introduces the deployment methods of DeepFlow. DeepFlow can be used to monitor container applications on multiple K8s, and cloud host applications in multiple VPCs. The content of this chapter is arranged as follows: +This chapter introduces the deployment methods of DeepFlow. DeepFlow can be used to monitor container applications on multiple K8s and cloud host applications in multiple VPCs. The content of this chapter is arranged as follows: - [all-in-one](./all-in-one/): Quickly experience DeepFlow using a single virtual machine - [single-k8s](./single-k8s/): Deploy DeepFlow to monitor all applications on a K8s cluster, with all observability data automatically injected into `K8s resources` and `K8s custom labels` - [multi-k8s](./multi-k8s/): Deploy DeepFlow to monitor all applications across multiple K8s clusters - [legacy-host](./legacy-host/): Deploy DeepFlow to monitor all applications on traditional servers -- [cloud-host](./cloud-host/): Deploy DeepFlow to monitor all applications on cloud servers, with all observability data automatically injected into `cloud resources` labels +- [cloud-host](./cloud-host/): Deploy DeepFlow to monitor all applications on cloud servers, with all observability data automatically injected into `cloud resource` labels - [managed-k8s](./managed-k8s/): Deploy DeepFlow to monitor all applications on cloud provider-managed K8s clusters, with all observability data automatically injected into `cloud resources`, `K8s resources`, and `K8s custom labels` - [serverless-pod](./serverless-pod/): Deploy DeepFlow to monitor all applications within Serverless Pods - [upgrade](./upgrade/): DeepFlow upgrade @@ -23,79 +23,85 @@ If you currently do not have suitable resources to deploy DeepFlow, you can log - [Universal Service Map - Experience DeepFlow's AutoMetrics capability](../features/universal-map/auto-metrics/) - [Distributed Tracing - Experience DeepFlow's AutoTracing capability](../features/distributed-tracing/auto-tracing/) - [Eliminate Data Silos - Learn about DeepFlow's AutoTagging and SmartEncoding capabilities](../features/auto-tagging/eliminate-data-silos/) -- [Say Goodbye to High Baseline Troubles - Integrate Prometheus and other metric data](../integration/input/metrics/metrics-auto-tagging/) -- [Full-Stack Distributed Tracing - Integrate OpenTelemetry and other tracing data](../integration/input/tracing/full-stack-distributed-tracing/) +- [Say Goodbye to High Base Troubles - Integrate metrics data such as Prometheus](../integration/input/metrics/metrics-auto-tagging/) +- [Full-Stack Distributed Tracing - Integrate tracing data such as OpenTelemetry](../integration/input/tracing/full-stack-distributed-tracing/) # Running Permissions and Kernel Requirements The eBPF capabilities (AutoTracing, AutoProfiling) in DeepFlow have the following kernel version requirements: -| Architecture | Distribution | Kernel Version | kprobe | Golang uprobe | OpenSSL uprobe | perf | -| ------------ | --------------------- | ----------------- | ------ | ------------- | -------------- | ---- | -| X86 | CentOS 7.9 | 3.10.0 **[1]** | Y | Y **[2]** | Y **[2]** | Y | -| | RedHat 7.6 | 3.10.0 **[1]** | Y | Y **[2]** | Y **[2]** | Y | -| | \* | 4.9-4.13 | | | | Y | -| | \* | 4.14 **[3]** | Y | Y **[2]** | | Y | -| | \* | 4.15 | Y | Y **[2]** | | Y | +| Architecture | Distribution | Kernel Version | kprobe [1] | Golang uprobe | OpenSSL uprobe | perf | +| -------- | ------ | ------- | ---------- | ------------- | -------------- | ---- | +| X86 | CentOS 7.9 | 3.10.0-940+ **[2]** | Y | Y **[3]** | Y **[3]** | Y | +| | RedHat 7.6 | 3.10.0-940+ **[2]** | Y | Y **[3]** | Y **[3]** | Y | +| | \* | 4.14 **[4]** | Y | Y **[3]** | | Y | +| | \* | 4.15 | Y | Y **[3]** | | Y | | | \* | 4.16 | Y | Y | | Y | | | \* | 4.17+ | Y | Y | Y | Y | +| | SUSE 12 SP5 | 4.12 [5] | Y | Y | | Y | | ARM | CentOS 8 | 4.18 | Y | Y | Y | Y | | | EulerOS | 5.10+ | Y | Y | Y | Y | -| | KylinOS V10 SP2 | 4.19.90-25.24+ | Y | Y | Y | Y | +| | KylinOS V10 SP1 | 4.19.90-23 [6] | Y | Y | Y | Y | +| | KylinOS V10 SP2 | 4.19.90-25.24+ [7] | Y | Y | Y | Y | | | KylinOS V10 SP3 | 4.19.90-52.24+ | Y | Y | Y | Y | | | Other Distributions | 5.8+ | Y | Y | Y | Y | Additional notes on kernel versions: -- [1]: CentOS 7.9 and RedHat 7.6 have [backported some eBPF capabilities](https://www.redhat.com/en/blog/introduction-ebpf-red-hat-enterprise-linux-7) into the 3.10 kernel - - In these two distributions, the detailed kernel versions supported by DeepFlow are as follows ([dependent hook points](https://github.com/deepflowio/deepflow/blob/main/agent/src/ebpf/docs/probes-and-maps.md)): +- [1]: When Linux has BTF (BPF Type Format) enabled, and the kernel is greater than or equal to [5.5](https://github.com/torvalds/linux/commit/f1b9509c2fb0ef4db8d22dac9aef8e856a5d81f6) for X86 architecture, or greater than or equal to [6.0](https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git/commit/?h=linux-6.0.y&id=efc9909fdce00a827a37609628223cd45bf95d0b) for ARM architecture, the agent will automatically use fentry/fexit instead of kprobe/kretprobe, resulting in approximately a 15% performance improvement. +- [2]: CentOS 7.9 and RedHat 7.6 have [ported some eBPF capabilities](https://www.redhat.com/en/blog/introduction-ebpf-red-hat-enterprise-linux-7) into the 3.10 kernel. + - In these two distributions, the detailed kernel versions supported by DeepFlow are as follows ([dependent Hook points](https://github.com/deepflowio/deepflow/blob/main/agent/src/ebpf/docs/probes-and-maps.md)): - 3.10.0-957.el7.x86_64 - 3.10.0-1062.el7.x86_64 - 3.10.0-1127.el7.x86_64 - 3.10.0-1160.el7.x86_64 - Note RedHat's statement: - > The eBPF in Red Hat Enterprise Linux 7.6 is provided as Tech Preview and thus doesn't come with full support and is not suitable for deployment in production. It is provided with the primary goal to gain wider exposure, and potentially move to full support in the future. eBPF in Red Hat Enterprise Linux 7.6 is enabled only for tracing purposes, which allows attaching eBPF programs to probes, tracepoints and perf events. -- [2]: Golang/OpenSSL processes inside containers are not supported -- [3]: In kernel version 4.14, a `tracepoint` cannot be attached by multiple eBPF programs (e.g., two or more deepflow-agents cannot run simultaneously), this issue does not exist in other versions + > The eBPF in Red Hat Enterprise Linux 7.6 is provided as Tech Preview and thus doesn't come with full support and is not suitable for deployment in production. It is provided with the primary goal to gain wider exposure, and potentially move to full support in the future. eBPF in Red Hat Enterprise Linux 7.6 is enabled only for tracing purposes, which allows attaching eBPF programs to probes, tracepoints, and perf events. +- [3]: Golang/OpenSSL processes inside containers are not supported. +- [4]: In kernel version 4.14, a `tracepoint` cannot be attached by multiple eBPF programs (e.g., two or more deepflow-agents cannot run simultaneously), this issue does not exist in other versions. +- [5]: Currently supports SUSE 12 SP5 4.12.14, but the Linux community's 4.12 version still does not support it. +- [6]: Some kernels of KylinOS V10 SP1, such as 4.19.90-23.48.v2101.ky10.aarch64, run normally, but it is not guaranteed that all aarch64 architecture kernels of KylinOS V10 SP1 can run deepflow-agent normally. +- [7]: Some kernels of KylinOS V10 SP2, such as 4.19.90-24.4.v2101.ky10.aarch64, do not support `bpf_probe_read_user()` and cannot read any user-space data, thus not supporting AutoTracing functionality, but can support continuous profiling and file read/write tracing functions. Requirements for running permissions of deepflow-agent: - When running in a K8s environment, the permissions required to collect K8s information include: - `[Required]` Container permission: `HOST_PID` - `[Recommended]` Kernel permission: `SYS_ADMIN` - - Without this permission, the mapping relationship between rootns network interface names, Pod IPs, and Pod MACs will be obtained by parsing ARP/ICMPV6 packets, and WARN logs will be printed + - Without this permission, the mapping relationship between rootns network card name, Pod IP, and Pod MAC is obtained by parsing ARP/ICMPV6 packets, and a WARN log is printed. - `[Required]` Kernel permission: `SYS_PTRACE` - - `[Recommended]` File permission: Read-only access to the `/var/run/netns` directory - - Without this permission, the performance of obtaining container network namespaces will be affected - - deepflow-agent will prioritize obtaining container network namespaces from this directory - - If this directory cannot be accessed, the container network namespace will be obtained through `/proc/$pid/ns/net`, with two issues: - - The file will disappear when the process stops - - Different PIDs may correspond to the same namespace + - `[Recommended]` File permission: Read-only permission for the `/var/run/netns` directory + - Without this permission, the performance of obtaining the container network namespace will be affected. + - deepflow-agent will first obtain the container's network namespace from this directory. + - If this directory cannot be accessed, the container's network namespace is obtained through `/proc/$pid/ns/net`, with two issues: + - The process stops, and this file disappears. + - Different PIDs may correspond to the same namespace. - Permissions required to collect AF_PACKET traffic include: - `[Required]` Container permission: `HOST_NET` - `[Required]` Kernel permissions: `NET_RAW`, `NET_ADMIN` - `[Recommended]` Kernel permission: `IPC_LOCK` (including MAP_LOCKED, MAP_NORESERVE) - - Without this permission, cBPF performance will be significantly affected, and WARN logs will be printed + - Without this permission, cBPF performance will be significantly affected, and a WARN log will be printed. - Permissions required to collect eBPF data include: - `[Required]` System permission: `SELINUX = disabled` - `[Recommended]` Kernel permission: `SYS_ADMIN` - - Under kernel `Linux 5.8+`, `SYS_ADMIN` is not required, and a combination of `BPF` and `PERFMON` can be used instead - - Using `SYS_ADMIN` permission has no kernel `Linux 5.8+` version dependency + - Under kernel `Linux 5.8+`, `SYS_ADMIN` is not required, and a combination of `BPF` and `PERFMON` can be used instead. + - Using `SYS_ADMIN` permission has no kernel `Linux 5.8+` version dependency. - `[Required]` Kernel permissions: `SYS_RESOURCE`, `SYSLOG` - - `[Required]` File permission: Read-only access to the `/sys/kernel/debug/` directory - - Since the attach/detach operations of kprobe and uprobe type probes depend on the kernel debug subsystem, eBPF cannot be enabled without this permission - - Additionally, since this directory can only be accessed by the root user, the deepflow-agent process `must run as the root user` - - `[Required]` Ensure that the content of the file `/proc/sys/kernel/kptr_restrict` is not equal to 2, otherwise Continuous Profiler cannot be used - - When `kptr_restrict` is set to 2, all users cannot read kernel symbol addresses, even with `CAP_SYSLOG` permission - - The agent will check kernel symbol addresses at startup, and if they cannot be read, WARN logs will be printed - - Generally, this value defaults to 1; if it is set to 2, users can set the file content to 1 in advance - - `[Recommended]` File permission: Read-write access to the file `/proc/sys/net/core/bpf_jit_enable` - - When the content of this file is not equal to 1, eBPF performance will be significantly affected - - When the value is 1, deepflow-agent will read the value, and if it lacks read permission, WARN logs will be printed - - When the value is not 1, deepflow-agent will attempt to change it to 1, and if the modification fails, WARN logs will be printed - - If you want to achieve good eBPF performance without granting `write permission`, users can set the file content to 1 in advance - - In K8s, the deepflow-agent DaemonSet will by default enable a privileged init container to set this value to 1, allowing deepflow-agent to run in non-privileged mode - -The deepflow-agent requires the following get/list/watch permissions to call the K8s apiserver to synchronize information: + - `[Required]` File permission: Read-only permission for the `/sys/kernel/debug/` directory + - Since the attach/detach operations of kprobe and uprobe type probes depend on the kernel debug subsystem, eBPF cannot be enabled without this permission. + - At the same time, since this directory can only be accessed by the root user, the deepflow-agent process `must run as the root user`. + - `[Required]` Ensure that the content of the file `/proc/sys/kernel/kptr_restrict` is not equal to 2, otherwise Continuous Profiler cannot be used. + - When `kptr_restrict` is set to 2, kernel symbol addresses cannot be read, even with `CAP_SYSLOG` permission. + - The agent will check the value of `kptr_restrict` at startup, and if it cannot read the value, a WARN log will be printed. + - If the value is 2, all users cannot read kernel symbol addresses, even with `CAP_SYSLOG` permission. + - Generally, the default value of this parameter in systems is 1. If the value is 2, users can set the file content to 1 in advance. + - `[Recommended]` File permission: Read/write permission for the `/proc/sys/net/core/bpf_jit_enable` file + - When the content of this file is not equal to 1, eBPF performance will be significantly affected. + - When the value is 1, deepflow-agent will read this value, and if it cannot read it, a WARN log will be printed. + - When the value is not 1, deepflow-agent will try to modify it to 1, and if the modification fails, a WARN log will be printed. + - If users want to obtain good eBPF performance without granting `write permission`, they can set the file content to 1 in advance. + - The deepflow-agent DaemonSet in K8s will by default enable a privileged init container to set this value to 1, allowing the deepflow-agent to run in non-privileged mode. + +The deepflow-agent requires get/list/watch permissions for the following resources to synchronize information with the K8s apiserver: - `nodes` - `namespaces` @@ -110,7 +116,7 @@ The deepflow-agent requires the following get/list/watch permissions to call the - `ingresses` - `routes` -Additionally, associating K8s label information requires adaptation to the CNI. Currently, DeepFlow has adapted to the following CNIs: +Additionally, associating K8s label information requires adaptation to CNI. Currently, DeepFlow is adapted to the following CNIs: - Flannel - Calico @@ -123,4 +129,4 @@ Additionally, associating K8s label information requires adaptation to the CNI. - ACK Terway - QKE HostNIC - IPVlan -- MACVlan [additional configuration](../best-practice/special-environment-deployment/#macvlan) +- MACVlan [additional configuration](../best-practice/special-environment-deployment/#macvlan) \ No newline at end of file diff --git a/translate/translated/02-ce-install/02-all-in-one.md b/translate/translated/02-ce-install/02-all-in-one.md index b040ca28..6d8e96e5 100644 --- a/translate/translated/02-ce-install/02-all-in-one.md +++ b/translate/translated/02-ce-install/02-all-in-one.md @@ -9,7 +9,7 @@ permalink: /ce-install/all-in-one To facilitate installation and deployment, we provide two deployment methods for deepflow-server: Kubernetes and Docker Compose. In this chapter, we will start with an All-in-One DeepFlow and introduce how to deploy a DeepFlow experience environment. -# Deploying with Kubernetes +# Deploy Using Kubernetes ## Preparation @@ -19,7 +19,7 @@ To facilitate installation and deployment, we provide two deployment methods for ### Deploy All-in-One K8s -Use [sealos](https://github.com/labring/sealos) to quickly deploy a K8s cluster: +Quickly deploy a K8s cluster using [sealos](https://github.com/labring/sealos): ```bash # install sealos @@ -53,7 +53,7 @@ sealos run labring/helm:v3.8.2 ## Deploy All-in-One DeepFlow -Use Helm to install All-in-One DeepFlow: +Install the LTS version of All-in-One DeepFlow using Helm: ::: code-tabs#shell @@ -66,8 +66,7 @@ cat << EOF > values-custom.yaml global: allInOneLocalStorage: true EOF -helm install deepflow -n deepflow deepflow/deepflow --create-namespace \ - -f values-custom.yaml +helm install deepflow -n deepflow deepflow/deepflow --version 6.6.018 --create-namespace -f values-custom.yaml ``` @tab Use Aliyun @@ -81,19 +80,18 @@ global: image: repository: registry.cn-beijing.aliyuncs.com/deepflow-ce EOF -helm install deepflow -n deepflow deepflow/deepflow --create-namespace \ - -f values-custom.yaml +helm install deepflow -n deepflow deepflow/deepflow --version 6.6.018 --create-namespace -f values-custom.yaml ``` ::: Note: -- We recommend saving the contents of the helm `--set` parameter in a separate yaml file, refer to the [Advanced Configuration](../best-practice/server-advanced-config/) section. +- We recommend saving the contents of the helm `--set` parameter in a separate yaml file, as referenced in the [Advanced Configuration](../best-practice/server-advanced-config/) section. -## Access Grafana Page +## Access the Grafana Page -The output of the helm deployment of DeepFlow provides commands to get the URL and password for accessing Grafana. Example output: +The output from executing helm to deploy DeepFlow provides commands to obtain the URL and password for accessing Grafana. Example output: ```bash NODE_PORT=$(kubectl get --namespace deepflow -o jsonpath="{.spec.ports[0].nodePort}" services deepflow-grafana) @@ -108,7 +106,12 @@ Grafana URL: http://10.1.2.3:31999 Grafana auth: admin:deepflow ``` -# Deploying with Docker Compose +# Deploy Using Docker Compose + +We do not recommend using Docker to deploy the DeepFlow Server side for the following reasons: + +1. The Server side relies on K8s [lease](https://kubernetes.io/docs/concepts/architecture/leases/) for leader election, achieving high availability through multiple replicas. The Docker environment lacks this mechanism, causing the Server side to run only in single-replica mode. In scenarios with a large number of Agent nodes or high data collection volume, a single-replica instance may not handle high concurrent data volumes due to resource bottlenecks. +2. When the Server side is deployed in single-replica mode, the accompanying ClickHouse can only be deployed in a single-shard form, otherwise, it will lead to uneven data writing, which limits the data query speed to some extent. ## Preparation @@ -138,43 +141,53 @@ chmod +x $DOCKER_CONFIG/cli-plugins/docker-compose ## Deploy All-in-One DeepFlow -Set the environment variable DOCKER_HOST_IP to the IP of the physical network card of the machine +Download the DeepFlow docker-compose package ```bash -unset DOCKER_HOST_IP -DOCKER_HOST_IP="10.1.2.3" # FIXME: Deploy the environment machine IP +wget https://deepflow-ce.oss-cn-beijing.aliyuncs.com/pkg/docker-compose/latest/linux/deepflow-docker-compose.tar +tar -zxf deepflow-docker-compose.tar ``` -Download and install All-in-One DeepFlow +Configure the `.env` variables + +```bash +vim ./deepflow-docker-compose/.env +DEEPFLOW_VERSION=v6.6 # FIXME: DeepFlow Version +NODE_IP_FOR_DEEPFLOW=192.168.101.116 # FIXME: Node IP +``` + +Install DeepFlow ```bash -wget https://deepflow-ce.oss-cn-beijing.aliyuncs.com/pkg/docker-compose/latest/linux/deepflow-docker-compose.tar -tar -zxf deepflow-docker-compose.tar -sed -i "s|FIX_ME_ALLINONE_HOST_IP|$DOCKER_HOST_IP|g" deepflow-docker-compose/docker-compose.yaml docker compose -f deepflow-docker-compose/docker-compose.yaml up -d ``` ## Deploy DeepFlow Agent -Refer to [Monitoring Traditional Servers](./legacy-host) to deploy deepflow-agent for this server. +Refer to [Monitor Traditional Servers](./legacy-host) to deploy deepflow-agent on this server. -## Access Grafana Page +## Access the Grafana Page -The port for DeepFlow Grafana deployed using Docker Compose is 3000, and the user password is admin:deepflow. +After deploying through docker compose, point your browser to `http://<$NODE_IP_FOR_DEEPFLOW>:3000` to log in to the Grafana console. -For example, if the machine's IP is 10.1.2.3, the Grafana access URL is http://10.1.2.3:3000 +Default credentials: -## Limitations - -- In this deployment mode, both deepflow-server and clickhouse do not support horizontal scaling. -- Since some capabilities of deepflow-server rely on Kubernetes, the docker-compose deployment mode cannot monitor cloud servers. You can refer to [Monitoring Traditional Servers](./legacy-host) to monitor cloud hosts. +- Username: admin +- Password: deepflow # Download deepflow-ctl -deepflow-ctl is a command-line tool for managing DeepFlow. It is recommended to download it to the K8s Node where deepflow-server is located for subsequent use: +deepflow-ctl is the command-line management tool for DeepFlow. It is recommended to deploy it on the K8s Node where deepflow-server is located for subsequent use in Agent [group configuration management](../best-practice/agent-advanced-config) and other maintenance operations: ```bash -curl -o /usr/bin/deepflow-ctl https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/stable/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl +# Sync with the current server version +Version=v6.6 + +# Download using variables +curl -o /usr/bin/deepflow-ctl \ + "https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/$Version/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl" + +# Add execution permissions chmod a+x /usr/bin/deepflow-ctl ``` @@ -182,6 +195,6 @@ chmod a+x /usr/bin/deepflow-ctl - [Universal Service Map - Experience DeepFlow's AutoMetrics Capability](../features/universal-map/auto-metrics/) - [Distributed Tracing - Experience DeepFlow's AutoTracing Capability](../features/distributed-tracing/auto-tracing/) -- [Eliminate Data Silos - Learn about DeepFlow's AutoTagging and SmartEncoding Capabilities](../features/auto-tagging/eliminate-data-silos/) -- [Say Goodbye to High Baseline Troubles - Integrate Prometheus and other metric data](../integration/input/metrics/metrics-auto-tagging/) -- [Full-Stack Distributed Tracing - Integrate OpenTelemetry and other tracing data](../integration/input/tracing/full-stack-distributed-tracing/) +- [Eliminate Data Silos - Learn About DeepFlow's AutoTagging and SmartEncoding Capabilities](../features/auto-tagging/eliminate-data-silos/) +- [Say Goodbye to High Base Troubles - Integrate Metrics Data Such as Prometheus](../integration/input/metrics/metrics-auto-tagging/) +- [Full-Stack Distributed Tracing - Integrate Tracing Data Such as OpenTelemetry](../integration/input/tracing/full-stack-distributed-tracing/) \ No newline at end of file diff --git a/translate/translated/02-ce-install/03-single-k8s.md b/translate/translated/02-ce-install/03-single-k8s.md index 44d2dc5f..35899cb7 100644 --- a/translate/translated/02-ce-install/03-single-k8s.md +++ b/translate/translated/02-ce-install/03-single-k8s.md @@ -7,9 +7,7 @@ permalink: /ce-install/single-k8s # Introduction -If you have deployed applications in a K8s cluster, this chapter introduces how to use DeepFlow for monitoring. -DeepFlow can collect observability signals (AutoMetrics, AutoTracing, AutoProfiling) from all Pods with zero intrusion, -and automatically inject `K8s resources` and `K8s custom labels` tags (AutoTagging) into all observability data based on information obtained from the apiserver. +If you have deployed applications in a K8s cluster, this chapter explains how to use DeepFlow for monitoring. DeepFlow can collect observability signals (AutoMetrics, AutoTracing, AutoProfiling) from all Pods without interference and automatically injects `K8s resources` and `K8s custom labels` tags (AutoTagging) into all observability data based on information obtained from the apiserver. # Preparation @@ -46,10 +44,9 @@ end ## Storage Class -We recommend using Persistent Volumes to store MySQL and ClickHouse data to avoid unnecessary maintenance costs. -You can provide a default Storage Class or add the `--set global.storageClass=` parameter to select a Storage Class for creating PVC. +We recommend using Persistent Volumes to store data for MySQL and ClickHouse to avoid unnecessary maintenance costs. You can provide a default Storage Class or add the parameter `--set global.storageClass=` to select a Storage Class for creating PVC. -You can choose [OpenEBS](https://openebs.io/) to create PVC: +You may choose [OpenEBS](https://openebs.io/) for creating PVC: ```bash kubectl apply -f https://openebs.github.io/charts/openebs-operator.yaml @@ -68,7 +65,7 @@ Install DeepFlow using Helm: ```bash helm repo add deepflow https://deepflowio.github.io/deepflow helm repo update deepflow # use `helm repo update` when helm < 3.7.0 -helm install deepflow -n deepflow deepflow/deepflow --create-namespace +helm install deepflow -n deepflow deepflow/deepflow --version 6.6.018 --create-namespace ``` @tab Use Aliyun @@ -81,8 +78,7 @@ global: image: repository: registry.cn-beijing.aliyuncs.com/deepflow-ce EOF -helm install deepflow -n deepflow deepflow/deepflow --create-namespace \ - -f values-custom.yaml +helm install deepflow -n deepflow deepflow/deepflow --version 6.6.018 --create-namespace -f values-custom.yaml ``` ::: @@ -91,20 +87,27 @@ Note: - Use helm --set global.storageClass to specify the storageClass - Use helm --set global.replicas to specify the number of replicas for deepflow-server and clickhouse -- We recommend saving the contents of the helm `--set` parameters in a separate yaml file, refer to the [Advanced Configuration](../best-practice/server-advanced-config/) section. +- We recommend saving the contents of helm's `--set` parameters in a separate yaml file, refer to the [Advanced Configuration](../best-practice/server-advanced-config/) section. # Download deepflow-ctl deepflow-ctl is a command-line tool for managing DeepFlow. It is recommended to download it to the K8s Node where deepflow-server is located for subsequent use: ```bash -curl -o /usr/bin/deepflow-ctl https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/stable/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl +# Set temporary variable +Version=v6.6 + +# Download using the variable +curl -o /usr/bin/deepflow-ctl \ + "https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/$Version/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl" + +# Add execute permission chmod a+x /usr/bin/deepflow-ctl ``` # Access the Grafana Page -The output of the helm deployment of DeepFlow provides commands to get the URL and password for accessing Grafana. Example output: +The output from executing helm to deploy DeepFlow provides commands to obtain the URL and password for accessing Grafana. Example output: ```bash NODE_PORT=$(kubectl get --namespace deepflow -o jsonpath="{.spec.ports[0].nodePort}" services deepflow-grafana) @@ -112,7 +115,7 @@ NODE_IP=$(kubectl get nodes -o jsonpath="{.items[0].status.addresses[0].address} echo -e "Grafana URL: http://$NODE_IP:$NODE_PORT \nGrafana auth: admin:deepflow" ``` -Example output after executing the above commands: +Example output after executing the above command: ```text Grafana URL: http://10.1.2.3:31999 @@ -121,8 +124,8 @@ Grafana auth: admin:deepflow # Next Steps -- [Universal Service Map - Experience DeepFlow's AutoMetrics capability](../features/universal-map/auto-metrics/) -- [Distributed Tracing - Experience DeepFlow's AutoTracing capability](../features/distributed-tracing/auto-tracing/) -- [Eliminate Data Silos - Learn about DeepFlow's AutoTagging and SmartEncoding capabilities](../features/auto-tagging/eliminate-data-silos/) -- [Say Goodbye to High Latency - Integrate metrics data like Prometheus](../integration/input/metrics/metrics-auto-tagging/) -- [Full-Stack Distributed Tracing - Integrate tracing data like OpenTelemetry](../integration/input/tracing/full-stack-distributed-tracing/) +- [Universal Service Map - Experience DeepFlow's AutoMetrics Capability](../features/universal-map/auto-metrics/) +- [Distributed Tracing - Experience DeepFlow's AutoTracing Capability](../features/distributed-tracing/auto-tracing/) +- [Eliminate Data Silos - Learn About DeepFlow's AutoTagging and SmartEncoding Capabilities](../features/auto-tagging/eliminate-data-silos/) +- [Say Goodbye to High Base Troubles - Integrate Metrics Data like Prometheus](../integration/input/metrics/metrics-auto-tagging/) +- [Full-Stack Distributed Tracing - Integrate Tracing Data like OpenTelemetry](../integration/input/tracing/full-stack-distributed-tracing/) \ No newline at end of file diff --git a/translate/translated/02-ce-install/04-multi-k8s.md b/translate/translated/02-ce-install/04-multi-k8s.md index cd9d6032..e4a9f18e 100644 --- a/translate/translated/02-ce-install/04-multi-k8s.md +++ b/translate/translated/02-ce-install/04-multi-k8s.md @@ -7,9 +7,9 @@ permalink: /ce-install/multi-k8s # Introduction -DeepFlow Server can serve DeepFlow Agents in multiple K8s clusters. Assuming you have already deployed DeepFlow Server in one K8s cluster, this chapter explains how to monitor other K8s clusters. +DeepFlow Server can serve DeepFlow Agents from multiple K8s clusters. Assuming you have already deployed DeepFlow Server in one K8s cluster, this chapter explains how to monitor other K8s clusters. -# Preparation +# Preparations ## Deployment Topology @@ -30,11 +30,11 @@ end ## Ensure Different K8s Clusters Can Be Distinguished -DeepFlow uses the MD5 value of the K8s CA file to distinguish different clusters. Please check the `/run/secrets/kubernetes.io/serviceaccount/ca.crt` file in the Pods of different K8s clusters to ensure that the CA files of different clusters are different. +DeepFlow uses the MD5 value of the K8s CA file to distinguish between different clusters. Please check the `/run/secrets/kubernetes.io/serviceaccount/ca.crt` file in the Pods of different K8s clusters to ensure that the CA files are different. -If your different K8s clusters use the same CA file, you need to use `deepflow-ctl domain create` to obtain a `K8sClusterID` before deploying deepflow-agent in multiple clusters: +If your different K8s clusters use the same CA file, before deploying deepflow-agent in multiple clusters, you need to use `deepflow-ctl domain create` to create a `k8s domain` and obtain its `$CLUSTER_NAME` and `$CLUSTER_ID`: -Note: It is uncommon for multiple K8s clusters to have the same CA file. Nevertheless, we still recommend manually connecting the deepflow-agent of other K8s clusters to the deepflow-server cluster. The advantage of manual connection is that you can customize the K8s cluster name displayed in the Grafana dashboard. You can create a custom K8s cluster domain using `deepflow-ctl domain create -f custom-domain.yaml`: +Note: It is uncommon for multiple K8s clusters to have the same CA file. Nevertheless, we still recommend manually connecting the deepflow-agent of other K8s clusters to the deepflow-server cluster. The advantage of manual connection is that you can customize the K8s cluster name displayed in the Grafana Dashboard. You can create a custom K8s cluster domain via `deepflow-ctl domain create -f custom-domain.yaml`: ```bash # Name (you can customize the cluster name, for example, beijing-prod-k8s) @@ -42,7 +42,7 @@ name: $CLUSTER_NAME # FIXME # Type of cloud platform type: kubernetes config: - ## Regional identifier (must use this default value) + ## Regional identifier (must use this default valued) #region_uuid: ffffffff-ffff-ffff-ffff-ffffffffffff ## Resource synchronization controller (it is recommended to use the default setting here) #controller_ip: 127.0.0.1 @@ -61,7 +61,7 @@ deepflow-ctl domain list $CLUSTER_NAME # Deploy deepflow-agent -Use Helm to install deepflow-agent. If the service used by deepflow-server is the default NodePort type, directly fill in the deepflow-server Node IP under `deepflowServerNodeIPS`; if the [service used by deepflow-server is of LoadBalancer type](../best-practice/production-deployment/#优化-deepflow-agent-到-deepflow-server-的流量路径), directly fill in the LoadBalancer VIP. +Use Helm to install deepflow-agent. If the service used by deepflow-server is the default NodePort type, fill in the deepflow-server Node IP directly under `deepflowServerNodeIPS`; if the [service used by deepflow-server is of LoadBalancer type](../best-practice/production-deployment/#优化-deepflow-agent-到-deepflow-server-的流量路径), then directly fill in the LoadBalancer VIP. ::: code-tabs#shell @@ -78,8 +78,7 @@ EOF helm repo add deepflow https://deepflowio.github.io/deepflow helm repo update deepflow # use `helm repo update` when helm < 3.7.0 -helm install deepflow-agent -n deepflow deepflow/deepflow-agent --create-namespace \ - -f values-custom.yaml +helm install deepflow-agent -n deepflow deepflow/deepflow-agent --version 6.6.018 --create-namespace -f values-custom.yaml ``` @tab Use Aliyun @@ -97,18 +96,17 @@ EOF helm repo add deepflow https://deepflow-ce.oss-cn-beijing.aliyuncs.com/chart/stable helm repo update deepflow # use `helm repo update` when helm < 3.7.0 -helm install deepflow-agent -n deepflow deepflow/deepflow-agent --create-namespace \ - -f values-custom.yaml +helm install deepflow-agent -n deepflow deepflow/deepflow-agent --version 6.6.018 --create-namespace -f values-custom.yaml ``` ::: -We recommend configuring the `deepflowServerNodeIPS` of deepflow-agent to one or more relatively fixed Node IPs of the K8s cluster during the above deployment process. +We recommend configuring `deepflowServerNodeIPS` for deepflow-agent in the above deployment process to one or more relatively fixed Node IPs of the K8s cluster. # Next Steps -- [Universal Service Map - Experience DeepFlow's AutoMetrics Capability](../features/universal-map/auto-metrics/) -- [Distributed Tracing - Experience DeepFlow's AutoTracing Capability](../features/distributed-tracing/auto-tracing/) -- [Eliminate Data Silos - Learn About DeepFlow's AutoTagging and SmartEncoding Capabilities](../features/auto-tagging/eliminate-data-silos/) -- [Say Goodbye to High Cardinality Issues - Integrate Metrics Data from Prometheus, etc.](../integration/input/metrics/metrics-auto-tagging/) -- [Full-Stack Distributed Tracing - Integrate Tracing Data from OpenTelemetry, etc.](../integration/input/tracing/full-stack-distributed-tracing/) +- [Universal Service Map - Experience DeepFlow's AutoMetrics capability](../features/universal-map/auto-metrics/) +- [Distributed Tracing - Experience DeepFlow's AutoTracing capability](../features/distributed-tracing/auto-tracing/) +- [Eliminate Data Silos - Learn about DeepFlow's AutoTagging and SmartEncoding capabilities](../features/auto-tagging/eliminate-data-silos/) +- [Say Goodbye to High Cardinality Issues - Integrate Prometheus and other metrics data](../integration/input/metrics/metrics-auto-tagging/) +- [Full-Stack Distributed Tracing - Integrate OpenTelemetry and other tracing data](../integration/input/tracing/full-stack-distributed-tracing/) \ No newline at end of file diff --git a/translate/translated/02-ce-install/05-legacy-host.md b/translate/translated/02-ce-install/05-legacy-host.md index 4edfbf34..d9085220 100644 --- a/translate/translated/02-ce-install/05-legacy-host.md +++ b/translate/translated/02-ce-install/05-legacy-host.md @@ -7,7 +7,7 @@ permalink: /ce-install/legacy-host # Introduction -DeepFlow supports monitoring legacy servers. Note that DeepFlow Server must run on K8s. If you do not have a K8s cluster, you can refer to the [All-in-One Quick Deployment](./all-in-one/) section to deploy DeepFlow Server first. +DeepFlow supports monitoring legacy servers. Note that the DeepFlow Server must run on top of K8s. If you do not have a K8s cluster, you can refer to the [All-in-One Quick Deployment](./all-in-one/) section to deploy the DeepFlow Server first. # Deployment Topology @@ -31,9 +31,9 @@ end # Configure DeepFlow Server -## Update deepflow-server Configuration +## Update deepflow-server configuration -Check if all network segments of the server are in the following list of network segments +Check whether all network segments of the server are included in the following list: ```yaml local_ip_ranges: @@ -44,9 +44,10 @@ local_ip_ranges: - 224.0.0.0-240.255.255.255 ``` -If not, you need to add the missing server network segments to the `local_ip_ranges` list in the custom configuration file below. For example, if the host IP is 100.42.32.213, you need to add the corresponding 100.42.32.0/24 network segment to the configuration. +If not, you need to add the missing server network segments to the `local_ip_ranges` list in the custom configuration file below. +For example: if the host IP is 100.42.32.213, you need to add the corresponding 100.42.32.0/24 segment to the configuration. -Modify the `values-custom.yaml` custom configuration file: +Edit the `values-custom.yaml` custom configuration file: ```yaml configmap: @@ -64,7 +65,7 @@ configmap: trident-type-for-unkonw-vtap: 3 # required ``` -Update deepflow +Update deepflow: ```bash helm upgrade deepflow -n deepflow -f values-custom.yaml deepflow/deepflow @@ -74,7 +75,7 @@ kubectl delete pods -n deepflow -l app=deepflow -l component=deepflow-server ## Create Host Domain -Just like monitoring multiple K8s clusters requires creating a K8s domain, here you also need to create a domain specifically for synchronizing servers. +Just like when monitoring multiple K8s clusters you need to create a K8s domain, here you also need to create a domain specifically for synchronizing servers. ```bash unset DOMAIN_NAME @@ -95,14 +96,15 @@ unset AGENT_GROUP AGENT_GROUP="legacy-host" # FIXME: domain name deepflow-ctl agent-group create $AGENT_GROUP -deepflow-ctl agent-group list $AGENT_GROUP # Get agent-group ID +deepflow-ctl agent-group list $AGENT_GROUP # get agent-group-id ``` -Create the agent group configuration file `agent-group-config.yaml`, specify `vtap_group_id` and enable `platform_enabled` to allow deepflow-agent to synchronize the server's network information to deepflow-server. +Use [deepflow-ctl](../best-practice/agent-advanced-config) to create an agent group configuration, so that deepflow-agent can send server network information to deepflow-server in self-synchronization mode. ```yaml -vtap_group_id: g-ffffff # FIXME -platform_enabled: 1 +inputs: + resources: + workload_resource_sync_enabled: true ``` Create the agent group configuration: @@ -113,14 +115,15 @@ deepflow-ctl agent-group-config create -f agent-group-config.yaml # Deploy DeepFlow Agent -Download deepflow-agent +Note: The deepflow-agent version must be ≤ the deepflow-server version, otherwise registration and data reporting issues may occur. ::: code-tabs#shell @tab rpm ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/rpm/agent/stable/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-rpm.zip +AGENT_VERSION=v6.6 FIXME: Keep this in sync with the server version +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/rpm/agent/$AGENT_VERSION/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-rpm.zip unzip deepflow-agent-rpm.zip yum -y localinstall x86_64/deepflow-agent-1.0*.rpm ``` @@ -128,7 +131,8 @@ yum -y localinstall x86_64/deepflow-agent-1.0*.rpm @tab deb ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/deb/agent/stable/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-deb.zip +AGENT_VERSION=v6.6 FIXME: Keep this in sync with the server version +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/deb/agent/$AGENT_VERSION/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-deb.zip unzip deepflow-agent-deb.zip dpkg -i x86_64/deepflow-agent-1.0*.systemd.deb ``` @@ -136,7 +140,8 @@ dpkg -i x86_64/deepflow-agent-1.0*.systemd.deb @tab binary file ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/agent/stable/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent.tar.gz +AGENT_VERSION=v6.6 FIXME: Keep this in sync with the server version +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/agent/$AGENT_VERSION/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent.tar.gz tar -zxvf deepflow-agent.tar.gz -C /usr/sbin/ cat << EOF > /etc/systemd/system/deepflow-agent.service @@ -171,7 +176,7 @@ services: image: registry.cn-hongkong.aliyuncs.com/deepflow-ce/deepflow-agent:v6.5 container_name: deepflow-agent restart: always - #privileged: true ## Docker version below 20.10.10 requires the opening of the privileged mode, See https://github.com/moby/moby/pull/42836 + #privileged: true ## Docker version below 20.10.10 requires enabling privileged mode, See https://github.com/moby/moby/pull/42836 cap_add: - SYS_ADMIN - SYS_RESOURCE @@ -193,12 +198,12 @@ docker compose -f deepflow-agent-docker-compose.yaml up -d ::: -Modify the deepflow-agent configuration file `/etc/deepflow-agent.yaml`: +Edit the deepflow-agent configuration file `/etc/deepflow-agent.yaml`: ```yaml controller-ips: - 10.1.2.3 # FIXME: K8s Node IPs -vtap-group-id-request: 'g-fffffff' # FIXME: agent-group ID +vtap-group-id-request: 'g-fffffff' # FIXME: ``` Start deepflow-agent: @@ -210,13 +215,16 @@ systemctl restart deepflow-agent **Note**: -If deepflow-agent cannot start normally due to missing dependencies, you can download the statically linked compiled deepflow-agent. Note that the statically linked compiled deepflow-agent has severe performance issues under multithreading: +If deepflow-agent fails to start due to missing dependency libraries, you can download the statically linked compiled deepflow-agent. +Be aware that the statically linked compiled deepflow-agent has serious performance issues under multithreading: + ::: code-tabs#shell @tab rpm ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/rpm/agent/stable/linux/static-link/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-rpm.zip +AGENT_VERSION=v6.6 FIXME: Keep this in sync with the server version +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/rpm/agent/$AGENT_VERSION/linux/static-link/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-rpm.zip unzip deepflow-agent-rpm.zip yum -y localinstall x86_64/deepflow-agent-1.0*.rpm ``` @@ -224,7 +232,8 @@ yum -y localinstall x86_64/deepflow-agent-1.0*.rpm @tab deb ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/deb/agent/stable/linux/static-link/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-deb.zip +AGENT_VERSION=v6.6 FIXME: Keep this in sync with the server version +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/deb/agent/$AGENT_VERSION/linux/static-link/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-deb.zip unzip deepflow-agent-deb.zip dpkg -i x86_64/deepflow-agent-1.0*.systemd.deb ``` @@ -232,7 +241,8 @@ dpkg -i x86_64/deepflow-agent-1.0*.systemd.deb @tab binary file ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/agent/stable/linux/static-link/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent.tar.gz +AGENT_VERSION=v6.6 FIXME: Keep this in sync with the server version +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/agent/$AGENT_VERSION/linux/static-link/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent.tar.gz tar -zxvf deepflow-agent.tar.gz -C /usr/sbin/ cat << EOF > /etc/systemd/system/deepflow-agent.service @@ -259,8 +269,8 @@ systemctl daemon-reload # Next Steps -- [Universal Service Map - Experience DeepFlow's AutoMetrics Capability](../features/universal-map/auto-metrics/) -- [Distributed Tracing - Experience DeepFlow's AutoTracing Capability](../features/distributed-tracing/auto-tracing/) -- [Eliminate Data Silos - Learn About DeepFlow's AutoTagging and SmartEncoding Capabilities](../features/auto-tagging/eliminate-data-silos/) -- [Say Goodbye to High Baseline Troubles - Integrate Metrics Data from Prometheus, etc.](../integration/input/metrics/metrics-auto-tagging/) -- [Full-Stack Distributed Tracing - Integrate Tracing Data from OpenTelemetry, etc.](../integration/input/tracing/full-stack-distributed-tracing/) \ No newline at end of file +- [Universal Service Map - Experience DeepFlow's AutoMetrics capability](../features/universal-map/auto-metrics/) +- [Distributed Tracing - Experience DeepFlow's AutoTracing capability](../features/distributed-tracing/auto-tracing/) +- [Eliminate Data Silos - Learn about DeepFlow's AutoTagging and SmartEncoding capabilities](../features/auto-tagging/eliminate-data-silos/) +- [Say Goodbye to High Cardinality Issues - Integrate Prometheus and other metrics data](../integration/input/metrics/metrics-auto-tagging/) +- [Full-Stack Distributed Tracing - Integrate OpenTelemetry and other tracing data](../integration/input/tracing/full-stack-distributed-tracing/) \ No newline at end of file diff --git a/translate/translated/02-ce-install/08-serverless-pod.md b/translate/translated/02-ce-install/08-serverless-pod.md index bc1b8036..791da343 100644 --- a/translate/translated/02-ce-install/08-serverless-pod.md +++ b/translate/translated/02-ce-install/08-serverless-pod.md @@ -7,7 +7,7 @@ permalink: /ce-install/serverless-pod # Introduction -DeepFlow Agent can be deployed as a Sidecar within a Serverless Pod. Assuming you have already deployed DeepFlow Server in a K8s cluster, this chapter explains how to monitor applications within a Serverless Pod. +DeepFlow Agent can be deployed as a Sidecar inside a Serverless Pod. Assuming you have already deployed DeepFlow Server in a K8s cluster, this chapter describes how to monitor applications inside a Serverless Pod. # Deployment Topology @@ -28,27 +28,33 @@ end # Deploy deepflow-agent -Modify the value file to deploy deepflow-agent as a daemonset and inject it as a sidecar, then obtain the `clusterNAME` through `deepflow-ctl domain list`. +Modify the value file to deploy deepflow-agent as a daemonset and inject it as a sidecar, and obtain `clusterNAME` via `deepflow-ctl domain list`: ```bash cat << EOF > values-custom.yaml deployComponent: -- "daemonset" - "watcher" -tke_sidecar: true +- "daemonset" +tkeSidecar: true +daemonsetWatchDisabled: true clusterNAME: $clusterNAME # FIXME: domain name EOF -helm install deepflow-agent -n deepflow deepflow/deepflow-agent --create-namespace \ - -f values-custom.yaml +helm install deepflow-agent -n deepflow deepflow/deepflow-agent --version 6.6.018 --create-namespace -f values-custom.yaml ``` -If you do not want the sidecar form of deepflow-agent to take on the role of list-watch apiserver, it is recommended to deploy a separate deepflow-agent deployment to synchronize K8s resources. For specific methods, refer to [Deploy DeepFlow Agent in Deployment Mode](../best-practice/special-environment-deployment/#部署-deployment-模式-deepflow-agent). +The above command will deploy two sets of deepflow-agent: + +- watcher: A deepflow-agent deployment used to synchronize K8s resources. + - During deployment, the environment variable `K8S_WATCH_POLICY=watch-only` will be automatically injected. In this case, deepflow-agent will only synchronize K8s resources and will not collect observability data. +- daemonset: Injects deepflow-agent as a sidecar into each serverless pod to collect observability data. + - Note: When there is no watcher-type deepflow-agent running, deepflow-server will elect a daemonset-type deepflow-agent to synchronize K8s resources. + - Therefore, to ensure that such deepflow-agents do not consume more resources due to being elected for K8s resource synchronization, the environment variable `K8S_WATCH_POLICY=watch-disabled` is automatically injected for them. # Next Steps -- [Universal Service Map - Experience DeepFlow's AutoMetrics Capability](../features/universal-map/auto-metrics/) -- [Distributed Tracing - Experience DeepFlow's AutoTracing Capability](../features/distributed-tracing/auto-tracing/) -- [Eliminate Data Silos - Learn About DeepFlow's AutoTagging and SmartEncoding Capabilities](../features/auto-tagging/eliminate-data-silos/) -- [Say Goodbye to High Maintenance - Integrate Prometheus and Other Metric Data](../integration/input/metrics/metrics-auto-tagging/) -- [Full-Stack Distributed Tracing - Integrate OpenTelemetry and Other Tracing Data](../integration/input/tracing/full-stack-distributed-tracing/) +- [Universal Service Map - Experience DeepFlow's AutoMetrics capability](../features/universal-map/auto-metrics/) +- [Distributed Tracing - Experience DeepFlow's AutoTracing capability](../features/distributed-tracing/auto-tracing/) +- [Eliminate Data Silos - Learn about DeepFlow's AutoTagging and SmartEncoding capabilities](../features/auto-tagging/eliminate-data-silos/) +- [Say Goodbye to High Cardinality Issues - Integrate Prometheus and other metrics data](../integration/input/metrics/metrics-auto-tagging/) +- [Full-Stack Distributed Tracing - Integrate OpenTelemetry and other tracing data](../integration/input/tracing/full-stack-distributed-tracing/) \ No newline at end of file diff --git a/translate/translated/02-ce-install/09-ai-agent.md b/translate/translated/02-ce-install/09-ai-agent.md index 1de75814..972a6f75 100644 --- a/translate/translated/02-ce-install/09-ai-agent.md +++ b/translate/translated/02-ce-install/09-ai-agent.md @@ -1,5 +1,5 @@ --- -title: Installing AskGPT Agent +title: Install AskGPT Agent permalink: /ce-install/ai-agent --- @@ -7,11 +7,123 @@ permalink: /ce-install/ai-agent # Prerequisites -Community edition of DeepFlow has been deployed in K8s. +When deploying DeepFlow, the AI component is not enabled by default. You need to manually add the AI component configuration in the `values-custom.yaml` file: -# Configuring Session Models +```yaml +stella-agent-ce: + enabled: true + replicas: 1 + hostNetwork: 'false' + dnsPolicy: ClusterFirst + imagePullSecrets: [] + nameOverride: '' + fullnameOverride: '' + podAnnotations: {} + + image: + repository: '{{ .Values.global.image.repository }}/deepflowio-stella-agent-ce' + pullPolicy: Always + # Overrides the image tag whose default is the chart appVersion. + tag: latest + + podSecurityContext: + {} + # fsGroup: 2000 + + securityContext: + # privileged: true + # capabilities: + # drop: + # - ALL + # readOnlyRootFilesystem: false + # runAsNonRoot: false + # runAsUser: 0 + + service: + ## Configuration for ClickHouse service + annotations: {} + labels: {} + clusterIP: '' + + ## Port for ClickHouse Service to listen on + ports: + - name: tcp + port: 20831 + targetPort: 20831 + nodePort: + protocol: TCP + # Additional ports to open for server service + additionalPorts: [] + externalIPs: [] + loadBalancerIP: '' + loadBalancerSourceRanges: [] + + ## Denotes if this Service desires to route external traffic to node-local or cluster-wide endpoints + externalTrafficPolicy: Cluster + type: ClusterIP + + readinessProbe: + httpGet: + path: /v1/health/ + port: http + failureThreshold: 10 + initialDelaySeconds: 15 + periodSeconds: 10 + successThreshold: 1 + livenessProbe: + failureThreshold: 6 + initialDelaySeconds: 15 + periodSeconds: 20 + successThreshold: 1 + httpGet: + path: /v1/health/ + port: http + timeoutSeconds: 1 + + configmap: + df-llm-agent.yaml: + daemon: true + api_timeout: 500 + sql_show: 'false' + log_file: '/var/log/df-llm-agent.log' + log_level: 'info' + instance_path: '/root/df-llm-agent' + mysql: + host: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.ip}}{{ else }}{{ $.Release.Name }}-mysql{{end}}' + port: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.port}}{{ else }}30130{{end}}' + user_name: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.username}}{{ else }}root{{end}}' + user_password: '{{ if $.Values.global.externalMySQL.enabled }}{{$.Values.global.externalMySQL.password}}{{ else }}{{ .Values.global.password.mysql }}{{end}}' + database: 'deepflow_llm' + + resources: + {} + # limits: + # cpu: 100m + # memory: 128Mi + # requests: + # cpu: 100m + # memory: 128Mi + + nodeSelector: {} + + tolerations: [] + + podAntiAffinityLabelSelector: [] + podAntiAffinityTermLabelSelector: [] + podAffinityLabelSelector: [] + podAffinityTermLabelSelector: [] + nodeAffinityLabelSelector: + [] + # - matchExpressions: + # - key: kubernetes.io/hostname + # operator: In + # values: controller + nodeAffinityTermLabelSelector: [] +``` + +# Configure Conversation Models -Currently, the service supports the following models, which can be enabled as needed through `values-custom.yaml` configuration: +Currently, the service supports the following models, which can be enabled as needed via `values-custom.yaml`: ```yaml stella-agent-ce: @@ -55,28 +167,28 @@ stella-agent-ce: Update DeepFlow: ```bash -helm upgrade deepflow -n deepflow -f values-custom.yaml deepflow/deepflow +helm upgrade deepflow -n deepflow -f values-custom.yaml deepflow/deepflow ``` # Using in Grafana -The AI model interpretation feature (alpha version) is currently available in `DeepFlow Topo Panel` and `DeepFlow Tracing Panel`: +The AI model interpretation feature (alpha version) is currently available in the `DeepFlow Topo Panel` and `DeepFlow Tracing Panel`: ![DeepFlow Topo Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024052966570a950a6ac.png) ![DeepFlow Tracing Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024052966570a93501df.png) -# Using the API +# Using via API Method: POST URL: - http[s]:://{ip}:{port}/v1/ai/stream/{platform_name}?engine={engine_name} -- Parameter Description - - ip/port: K8s Service of stella-agent, default port is 30831. - - platform_name: Name of the platform where the model is located, e.g., azure. - - engine_name: Name of the engine, e.g., DF-GPT4-32K. +- Parameter description: + - ip/port: The K8s Service of stella-agent, default port is 30831. + - platform_name: The name of the platform where the model is located, e.g., azure. + - engine_name: The engine name, e.g., DF-GPT4-32K. Header: @@ -88,72 +200,72 @@ Request JSON Body: ``` { - "system_content": "You are an expert in web development", # Model role positioning - "user_content": "How to optimize a web page" # Problem description + "system_content": "You are an expert in web technologies", # Model role definition + "user_content": "How to optimize a web page" # Question description } ``` Response: ```text -Web page optimization refers to improving the performance of a web page through various means, including load speed, response speed, and user experience. +Web page optimization refers to various methods to improve the performance of a web page, including load speed, response time, and user experience. -Web page optimization is a complex and comprehensive task that involves multiple aspects, including front-end, back-end, and network. It requires analysis and optimization based on specific situations. +Web page optimization is a complex and comprehensive task involving multiple aspects, including frontend, backend, and network. It requires analysis and optimization based on the specific situation. Here are some common web page optimization methods: -Front-end Optimization +Frontend Optimization -Reduce HTTP requests: Reducing HTTP requests can decrease network latency and browser parsing time. Methods to reduce HTTP requests include: -Combining CSS and JavaScript files. +Reduce HTTP requests: Reducing HTTP requests can decrease network latency and browser parsing time. This can be achieved by: +Merging CSS and JavaScript files. Using sprite images. Implementing lazy loading. -Optimize resources: Optimizing resources can reduce their size, thereby improving load speed. Methods to optimize resources include: +Optimize resources: Optimizing resources can reduce their size, thus improving load speed. This can be done by: Compressing images. Using appropriate image formats. Minifying CSS and JavaScript files. -Proper use of caching: Caching can reduce repeated network requests, thereby improving load speed. Methods to use caching properly include: +Use caching effectively: Caching can reduce repeated network requests, thus improving load speed. This can be done by: Setting appropriate cache headers. Using browser caching. -Optimize DOM structure: A reasonable DOM structure can reduce browser parsing and rendering time. Methods to optimize DOM structure include: -Reducing the number and levels of DOM elements. +Optimize DOM structure: A well-structured DOM can reduce browser parsing and rendering time. This can be done by: +Reducing the number and depth of DOM elements. Avoiding complex CSS selectors. -Optimize JavaScript code: Optimizing JavaScript code can improve code execution efficiency. Methods to optimize JavaScript code include: +Optimize JavaScript code: Optimizing JavaScript can improve execution efficiency. This can be done by: Avoiding global variables. Using caching. Using appropriate algorithms and data structures. -Back-end Optimization +Backend Optimization -Optimize database queries: Optimizing database queries can reduce the load on the database server, thereby improving page response speed. Methods to optimize database queries include: +Optimize database queries: Optimizing database queries can reduce the load on the database server, thus improving page response speed. This can be done by: Using appropriate indexes. Avoiding unnecessary queries. Using caching. -Optimize application server: Optimizing the application server can improve its performance. Methods to optimize the application server include: +Optimize application servers: Optimizing application servers can improve their performance. This can be done by: Using appropriate load balancing strategies. Using caching. Optimizing code. Network Optimization -Choose an appropriate CDN: A CDN can distribute content to servers closer to the user, reducing network latency. -Optimize DNS resolution: Optimizing DNS resolution can improve DNS resolution speed. Methods to optimize DNS resolution include: +Choose an appropriate CDN: A CDN can distribute content to servers closer to users, reducing network latency. +Optimize DNS resolution: Optimizing DNS resolution can improve DNS lookup speed. This can be done by: Using a CDN. Configuring DNS records. -Using Gzip compression: Gzip compression can reduce the amount of data transmitted, thereby improving load speed. +Use Gzip compression: Gzip compression can reduce the amount of data transmitted, thus improving load speed. Tools -Various tools can be used to test and analyze the performance of web pages, such as: +Various tools can be used to test and analyze web page performance, such as: Google PageSpeed Insights Lighthouse WebPageTest By using these tools, you can identify performance bottlenecks in web pages and perform corresponding optimizations. -Web page optimization is a continuous process that requires constant testing and optimization to achieve the best performance. +Web page optimization is an ongoing process that requires continuous testing and optimization to achieve the best performance. Here are some additional suggestions: Always consider performance when developing web pages. -Use performance testing tools to test the performance of web pages. -Monitor the performance of web pages and optimize them regularly. +Use performance testing tools to test web page performance. +Monitor web page performance and optimize regularly. I hope this information is helpful to you. ``` \ No newline at end of file diff --git a/translate/translated/02-ce-install/99-upgrade.md b/translate/translated/02-ce-install/99-upgrade.md index 5714241e..abaeb3b4 100644 --- a/translate/translated/02-ce-install/99-upgrade.md +++ b/translate/translated/02-ce-install/99-upgrade.md @@ -11,7 +11,7 @@ Upgrade DeepFlow to the latest version and obtain the latest Grafana dashboard. # Upgrade DeepFlow Server -Upgrade DeepFlow Server and the DeepFlow Agent in this cluster with a single Helm command: +Use Helm to upgrade DeepFlow Server and the DeepFlow Agent in this cluster with one command: ```bash helm repo update deepflow # use `helm repo update` when helm < 3.7.0 @@ -23,24 +23,31 @@ helm upgrade deepflow -n deepflow deepflow/deepflow -f values-custom.yaml Download the latest deepflow-ctl: ```bash -curl -o /usr/bin/deepflow-ctl https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/stable/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl +# Keep it in sync with the current server version +Version=v6.6 + +# Download using the variable +curl -o /usr/bin/deepflow-ctl \ + "https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/$Version/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl" + +# Add execute permission chmod a+x /usr/bin/deepflow-ctl ``` # Upgrade DeepFlow Agent -## Upgrade Agent in K8s Cluster +## Upgrade Agents in a K8s Cluster -Upgrade DeepFlow Agent with a single Helm command: +Use Helm to upgrade DeepFlow Agent with one command: ```bash helm repo update deepflow # use `helm repo update` when helm < 3.7.0 helm upgrade deepflow-agent -n deepflow deepflow/deepflow-agent -f values-custom.yaml ``` -## Remote Upgrade Agent on Cloud Servers +## Remotely Upgrade Agents on Cloud Servers -Upgrade DeepFlow Agent deployed on cloud servers and traditional servers using deepflow-ctl: +Use deepflow-ctl to upgrade DeepFlow Agents deployed on cloud servers and traditional servers: 1. Download the latest deepflow-agent: @@ -55,15 +62,15 @@ Upgrade DeepFlow Agent deployed on cloud servers and traditional servers using d deepflow-ctl repo agent create --arch x86 --image /usr/sbin/deepflow-agent ``` - If the same binary file name is uploaded multiple times, it will be overwritten; the uploaded binary will be compressed with a compression ratio of about 3.4. + If you upload a binary with the same file name multiple times, it will be overwritten; the uploaded binary will be compressed, with a compression ratio of about 3.4. -3. List the packages in the repository: +3. View the packages in the repository: ```bash deepflow-ctl repo agent list ``` -4. Execute the upgrade: +4. Perform the upgrade: ```bash OUTPUT=$(deepflow-ctl agent list | head -n 1) if [[ $OUTPUT == "VTAP_ID"* ]]; then @@ -79,13 +86,13 @@ Upgrade DeepFlow Agent deployed on cloud servers and traditional servers using d fi ``` -## Remote Upgrade K8s Agent +## Remotely Upgrade K8s Agents -Remote upgrading of agents can reduce operational steps and increase upgrade speed when upgrading multiple clusters in bulk. +Remote upgrade of Agents can reduce operation steps and improve upgrade speed when upgrading multiple clusters in batches. -K8s remote upgrade uses deepflow-agent to modify the permissions of the daemonset and configmap in its namespace, and modify its own daemonset parameters for remote upgrade. +K8s remote upgrade uses deepflow-agent in the cluster with permissions to modify the daemonset and configmap in its namespace, and modifies its own daemonset parameters for remote upgrade. -Add the image of the deepflow-agent to be upgraded. The --version-image in the command needs to point to the deepflow-agent binary file of the same version as the K8s image, so that deepflow-ctl can obtain the version information of the image. Ensure that the K8s cluster to be upgraded can correctly pull the image corresponding to --k8s-image. +Add the image of the deepflow-agent to be upgraded. The `--version-image` in the command should point to the deepflow-agent binary of the same version as the K8s image, so that deepflow-ctl can obtain the version information of the image. Make sure the K8s cluster to be upgraded can correctly pull the image specified by `--k8s-image`. ```bash deepflow-ctl repo agent create --arch x86 \ @@ -93,28 +100,28 @@ deepflow-ctl repo agent create --arch x86 \ --k8s-image registry.cn-beijing.aliyuncs.com/deepflow-ce/deepflowio-agent:latest ``` -Execute the remote upgrade command of deepflow-agent. Only one agent in a cluster needs to be specified to upgrade all agents in the cluster. +Execute the remote upgrade command for deepflow-agent. In a cluster, you only need to specify one agent to upgrade all agents in the cluster. ```bash deepflow-ctl agent-upgrade \ --image-name="kube.registry.local:5000/deepflow-agent:v6.4.4594" ``` -- **Note**: The current version does not support verifying the pullability and availability of the image. Please ensure that the image can be pulled, run, and the version is correct before use. -- **Note**: Only one deepflow-agent needs to be selected in a K8s cluster to trigger the upgrade. It will modify the configuration of the daemonset itself, so that all deepflow-agents are upgraded. -- **Note**: Be prepared to manually intervene to correct the image (due to reasons such as the image cannot be pulled or run). +- **Note**: The current version does not support verifying whether the image can be pulled and is available. Please ensure the image can be pulled, run, and has the correct version before use. +- **Note**: In a K8s cluster, only one deepflow-agent needs to be selected to trigger the upgrade. It will modify the daemonset configuration so that all deepflow-agents are upgraded. +- **Note**: Be prepared to manually intervene to fix the image (due to reasons such as the image being unavailable or unable to run). -# Obtain the Latest DeepFlow Grafana Dashboard +# Get the Latest DeepFlow Grafana Dashboard -Check if the image of the init container `init-grafana-ds-dh` of Grafana is `latest` and if the image pull policy is `Always`: +Check whether the image of Grafana's init container `init-grafana-ds-dh` is `latest` and whether the image pull policy is `Always`: ```bash kubectl get deployment -n deepflow deepflow-grafana -o yaml|grep -E 'image:|imagePullPolicy' ``` -If the image of the init container `init-grafana-ds-dh` of Grafana is not `latest` and the image pull policy is not `Always`, please modify them to `latest` and `Always`. +If the image of Grafana's init container `init-grafana-ds-dh` is not `latest` and the image pull policy is not `Always`, please change them to `latest` and `Always`. -Restart Grafana to pull the latest init container `init-grafana-ds-dh` image and obtain the latest dashboard: +Restart Grafana to pull the latest init container `init-grafana-ds-dh` image and get the latest dashboard: ```bash kubectl delete pods -n deepflow -l app.kubernetes.io/instance=deepflow -l app.kubernetes.io/name=grafana diff --git a/translate/translated/03-ee-install/01-saas/01-cloud.md b/translate/translated/03-ee-install/01-saas/01-cloud.md index 1a5eb845..36ad7a23 100644 --- a/translate/translated/03-ee-install/01-saas/01-cloud.md +++ b/translate/translated/03-ee-install/01-saas/01-cloud.md @@ -7,31 +7,31 @@ permalink: /ee-install/saas/cloud # Introduction -Registering cloud platforms on the DeepFlow web page and completing the integration with cloud platform APIs are prerequisites for the following DeepFlow functionalities: +Registering a cloud platform in the DeepFlow web interface and completing the integration with the cloud platform API is a prerequisite for the following DeepFlow features to function: -- Learning information about cloud server instances in public clouds to accept registration requests from DeepFlow Agents deployed within these cloud servers. -- Learning information about resources and tags such as VPCs, load balancers, and RDS in public clouds, and automatically injecting `cloud resource` tags into observability data collected by DeepFlow Agents. +- Learn public cloud server instance information to accept registration requests from DeepFlow Agents deployed inside cloud servers. +- Learn public cloud VPC, load balancer, RDS, and other resource and tag information, and automatically inject `cloud resource` tags into observability data collected by DeepFlow Agents. -This chapter will provide a detailed guide on how to register cloud platform information on the DeepFlow web page to complete the integration with cloud platform APIs. -Once registered, DeepFlow will automatically synchronize cloud resource information periodically through the APIs provided by the cloud platforms based on your configuration and build observability data tags for DeepFlow. +This section provides a detailed guide on how to register cloud platform information in the DeepFlow web interface to complete API integration with the cloud platform. +Once registered, DeepFlow will automatically synchronize cloud resource information periodically via the cloud platform’s API based on your configuration, and build observability data tags in DeepFlow. # Interaction Topology ![Interaction Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202407156694c77d6f050.jpeg) -# Supported Cloud Service Providers +# Supported Cloud Providers -DeepFlow currently supports API integration and cloud resource information synchronization with the following public clouds: -| Cloud Service Provider (English) | Cloud Service Provider (Chinese) | Type Identifier in DeepFlow | -| -------------------------------- | -------------------------------- | --------------------------- | -| AWS | AWS | aws | -| Aliyun | 阿里云 | aliyun | -| Baidu Cloud | 百度云 | baidu_bce | -| Huawei Cloud | 华为云 | huawei | -| Microsoft Azure | 微软云 | | -| QingCloud | 青云 | qingcloud | -| Tencent Cloud | 腾讯云 | tencent | -| Volcengine | 火山引擎 | volcengine | +DeepFlow currently supports API integration and cloud resource synchronization for the following public clouds: +| Cloud Provider (English) | Cloud Provider (Chinese) | Type Identifier in DeepFlow | +| ------------------------ | ------------------------ | --------------------------- | +| AWS | AWS | aws | +| Aliyun | 阿里云 | aliyun | +| Baidu Cloud | 百度云 | baidu_bce | +| Huawei Cloud | 华为云 | huawei | +| Microsoft Azure | 微软云 | | +| QingCloud | 青云 | qingcloud | +| Tencent Cloud | 腾讯云 | tencent | +| Volcengine | 火山引擎 | volcengine | # Aliyun @@ -39,48 +39,52 @@ DeepFlow currently supports API integration and cloud resource information synch 1. Go to `Resources` - `Resource Pool` - `Cloud Platform` 2. Click `New Cloud Platform` -3. Fill in the relevant cloud platform information and click `Confirm` to get a record of the cloud platform +3. Fill in the relevant cloud platform information and click `OK` to create a cloud platform record ![Register Cloud Platform (Aliyun)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080866b4a7076882c.png) ## Configuration Item Description -| Configuration Item | Content Example | Remarks | -| ------------------ | -------------------------------------- | ----------------------------------------------------------------------- | -| Cloud Platform Name| Example: `my-aliyun` | The name of the cloud platform as seen in DeepFlow, customizable | -| AccessKey ID | Example: `LTAI4FiU3ad3txLUSRg8xGfn` | Create an AccessKey in the Aliyun console and fill in the ID here | -| AccessKey Secret | Example: `itsHzkPo22jbtNZ61QEz3gc5bsPnXP` | Create an AccessKey in the Aliyun console and fill in the Secret here | -| Region Whitelist | Example: `South China 3 (Guangzhou), North China 6 (Ulanqab)` | List of regions where Aliyun ECS resources are located, separated by `, ` | +| Configuration Item | Example Value | Notes | +| ------------------- | --------------------------------------- | --------------------------------------------------------------------- | +| Cloud Platform Name | e.g., `my-aliyun` | The name of the cloud platform as displayed in DeepFlow, customizable | +| AccessKey ID | e.g., `LTAI4FiU3ad3txLUSRg8xGfn` | Create an AccessKey in the Aliyun console and enter the ID here | +| AccessKey Secret | e.g., `itsHzkPo22jbtNZ61QEz3gc5bsPnXP` | Create an AccessKey in the Aliyun console and enter the Secret here | +| Region Whitelist | e.g., `华南3(广州), 华北6(乌兰察布)` | List of regions where Aliyun ECS resources are located, separated by `, ` | ::: warning -The `Region Whitelist` must be filled in and must match the actual distribution of cloud server resources. If the `Region Whitelist` is empty (matching all regions) or too extensive, DeepFlow may query too many Aliyun regions, resulting in long query cycles. If the regions listed do not include the regions where your cloud servers are located, DeepFlow will not be able to learn the information of the cloud servers in those regions, and DeepFlow Agents will not be able to register. +`Region Whitelist` must be filled in and must match the actual distribution of your cloud server resources. +If the `Region Whitelist` is empty (matches all regions) or contains too many regions, DeepFlow may query too many Aliyun regions, resulting in long query cycles. +If the regions you enter do not include the regions where your cloud servers are located, DeepFlow will not be able to learn the cloud server information in those regions, and DeepFlow Agents will fail to register. ::: -**Steps to Create an AccessKey in the Aliyun Console**: +**Steps to create an AccessKey in the Aliyun console**: ![Create AccessKey in Aliyun Console](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240709668ce0a59992c.png) -**Steps to Query the Regions of Aliyun Resources**: +**Steps to check the regions where Aliyun resources are located**: -![Query Regions of Aliyun Resources](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240709668ce0a9687e6.png) +![Check Aliyun Resource Regions](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240709668ce0a9687e6.png) ## API Permission Description -DeepFlow will use the following APIs to learn resource information from Aliyun. If you need to restrict the resources that DeepFlow can learn, you can limit the resource permissions of the account used to generate the AccessKey in the Aliyun console: - -| API | Integration Content | Required | -| ----------------------------- | --------------------------- | -------- | -| DescribeRegions | Query region list | Required | -| DescribeVpcs | Query VPC list | Required | -| DescribeVSwitches | Query switch list | Required | -| DescribeInstances | Query cloud server instance list | Required | -| DescribeNetworkInterfaces | Query cloud server network interface list | Required | -| DescribeNatGateways | Query NAT gateway list | Optional | -| DescribeSnatTableEntries | Query NAT gateway SNAT rules | Optional | -| DescribeForwardTableEntries | Query NAT gateway DNAT rules | Optional | -| DescribeLoadBalancers | Query load balancers | Optional | -| DescribeLoadBalancerAttribute | Query load balancer listeners | Optional | -| DescribeHealthStatus | Query load balancer rules | Optional | +DeepFlow uses the following APIs to learn resource information from Aliyun. +If you need to restrict the resources DeepFlow can access, you can limit the permissions of the account used to generate the AccessKey in the Aliyun console: + +| Product | API | Permission | Integration Content | Required | +| ----------- | ----------------------------- | ------------------------- | --------------------------------- | -------- | +| Vpc | DescribeRegions | AliyunVPCReadOnlyAccess | Query region list | Yes | +| Vpc | DescribeVpcs | AliyunVPCReadOnlyAccess | Query VPC list | Yes | +| Vpc | DescribeVSwitches | AliyunVPCReadOnlyAccess | Query switch list | Yes | +| Ecs | DescribeInstances | AliyunECSReadOnlyAccess | Query cloud server instance list | Yes | +| Ecs | DescribeNetworkInterfaces | AliyunECSReadOnlyAccess | Query cloud server NIC list | Yes | +| Vpc | DescribeNatGateways | AliyunVPCReadOnlyAccess | Query NAT gateway list | No | +| Vpc | DescribeSnatTableEntries | AliyunVPCReadOnlyAccess | Query NAT gateway SNAT rules | No | +| Vpc | DescribeForwardTableEntries | AliyunVPCReadOnlyAccess | Query NAT gateway DNAT rules | No | +| Slb | DescribeLoadBalancers | AliyunSLBReadOnlyAccess | Query load balancers | No | +| Slb | DescribeLoadBalancerAttribute | AliyunSLBReadOnlyAccess | Query load balancer listeners | No | +| Slb | DescribeHealthStatus | AliyunSLBReadOnlyAccess | Query load balancer rules | No | +| Container Service | DescribeClusters | AliyunCSReadOnlyAccess | Query cluster list | No | # Tencent Cloud @@ -88,46 +92,49 @@ DeepFlow will use the following APIs to learn resource information from Aliyun. 1. Go to `Resources` - `Resource Pool` - `Cloud Platform` 2. Click `New Cloud Platform` -3. Fill in the relevant cloud platform information and click `Confirm` to get a record of the cloud platform +3. Fill in the relevant cloud platform information and click `OK` to create a cloud platform record ![Register Cloud Platform (Tencent Cloud)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080866b4a705b08bb.png) ## Configuration Item Description -| Configuration Item | Content Example | Remarks | -| ------------------ | -------------------------------------- | ----------------------------------------------------------------------- | -| Cloud Platform Name| Example: tencent-1 | The name of the cloud platform as seen in DeepFlow, customizable | -| AccessKey ID | Example: AKIDztZ0C9dHuIQJwKMeZEixykjTBhz2L | Fill in the `SecretId` generated after creating a new key in the `API Key Management` page under `Access Management` in the Tencent Cloud console (read-only permissions are sufficient) | -| AccessKey Secret | Example: itsHzkPo22jbtNZ61QEz3gc5bsPnXP | Fill in the `SecretKey` corresponding to the `SecretId` (read-only permissions are sufficient) | -| Region Whitelist | Example: East China (Shanghai) | List of regions where Tencent Cloud servers are located, multiple regions can be configured, regular expressions are not supported, regions are separated by `, ` | +| Configuration Item | Example Value | Notes | +| ------------------- | ------------------------------------- | ---------------------------------------------------------------------------------------------------------- | +| Cloud Platform Name | e.g., tencent-1 | The name of the cloud platform as displayed in DeepFlow, customizable | +| AccessKey ID | e.g., AKIDztZ0C9dHuIQJwKMeZEixykjTBhz2L | Enter the `SecretId` generated after creating a new key in Tencent Cloud `Access Management` - `API Key Management` (read-only permission is sufficient) | +| AccessKey Secret | e.g., itsHzkPo22jbtNZ61QEz3gc5bsPnXP | Enter the `SecretKey` corresponding to the `SecretId` (read-only permission is sufficient) | +| Region Whitelist | e.g., 华东地区(上海) | List of regions where Tencent Cloud servers are located, multiple regions can be configured, regex not supported, separated by `, ` | ::: warning -The `Region Whitelist` must be filled in and must match the actual distribution of cloud server resources. If the `Region Whitelist` is empty (matching all regions) or too extensive, DeepFlow may query too many Tencent Cloud regions, resulting in long query cycles. If the regions listed do not include the regions where your cloud servers are located, DeepFlow will not be able to learn the information of the cloud servers in those regions, and DeepFlow Agents will not be able to register. +`Region Whitelist` must be filled in and must match the actual distribution of your cloud server resources. +If the `Region Whitelist` is empty (matches all regions) or contains too many regions, DeepFlow may query too many Tencent Cloud regions, resulting in long query cycles. +If the regions you enter do not include the regions where your cloud servers are located, DeepFlow will not be able to learn the cloud server information in those regions, and DeepFlow Agents will fail to register. ::: ::: tip -The Tencent Cloud `Region` list includes: South China (Guangzhou), East China (Nanjing), North China (Beijing), Southwest China (Chengdu), Southwest China (Chongqing), Hong Kong, Macao, and Taiwan (Hong Kong, China), Northeast Asia (Seoul), Northeast Asia (Tokyo), Southeast Asia (Singapore), Southeast Asia (Bangkok), Southeast Asia (Jakarta), Western US (Silicon Valley), Europe (Frankfurt), South Asia (Mumbai), Eastern US (Virginia), South America (São Paulo), North America (Toronto) +Tencent Cloud `Region` list includes: 华南地区(广州), 华东地区(南京), 华北地区(北京), 西南地区(成都), 西南地区(重庆), 港澳台地区(中国香港), 亚太东北(首尔), 亚太东北(东京), 亚太东南(新加坡), 亚太东南(曼谷), 亚太东南(雅加达), 美国西部(硅谷), 欧洲地区(法兰克福), 亚太南部(孟买), 美国东部(弗吉尼亚), 南美地区(圣保罗), 北美地区(多伦多) ::: -**Steps to Create a Key in the Tencent Cloud Console**: +**Steps to create a key in the Tencent Cloud console**: ![Create Key in Tencent Cloud Console](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240719669a41be51191.png) ## API Permission Description -DeepFlow will use the following APIs to learn resource information from Tencent Cloud. If you need to restrict the resources that DeepFlow can learn, you can limit the resource permissions of the account used to generate the key in the Tencent Cloud console: - -| API | Integration Content | Required | -| ------------------------------------------------------ | ------------------------------- | -------- | -| DescribeRegions | Query region list | Required | -| DescribeZones | Query availability zone list | Required | -| DescribeVpcs | Query VPC list | Required | -| DescribeNatGateways | Query NAT gateways and related information | Required | -| DescribeNatGatewayDestinationIpPortTranslationNatRules | Query NAT gateway rules | Required | -| DescribeRouteTables | Query route tables | Required | -| DescribeSubnets | Query subnet list | Required | -| DescribeInstances | Query instance list | Required | -| DescribeNetworkInterfaces | Query elastic network interface list | Required | -| DescribeLoadBalancers | Query load balancer list | Required | -| DescribeListeners | Query load balancer listener list | Required | -| DescribeTargets | Query backend services bound to load balancers | Required | -| DescribeClassicalLBListeners | Query classical load balancer listener list | Required | \ No newline at end of file +DeepFlow uses the following APIs to learn resource information from Tencent Cloud. +If you need to restrict the resources DeepFlow can access, you can limit the permissions of the account used to generate the key in the Tencent Cloud console: + +| API | Integration Content | Required | +| ------------------------------------------------------ | ------------------------------------- | -------- | +| DescribeRegions | Query region list | Yes | +| DescribeZones | Query availability zone list | Yes | +| DescribeVpcs | Query VPC list | Yes | +| DescribeNatGateways | Query NAT gateways and related info | Yes | +| DescribeNatGatewayDestinationIpPortTranslationNatRules | Query NAT gateway rules | Yes | +| DescribeRouteTables | Query route tables | Yes | +| DescribeSubnets | Query subnet list | Yes | +| DescribeInstances | Query instance list | Yes | +| DescribeNetworkInterfaces | Query elastic NIC list | Yes | +| DescribeLoadBalancers | Query load balancer list | Yes | +| DescribeListeners | Query load balancer listener list | Yes | +| DescribeTargets | Query backend service list bound to load balancers | Yes | +| DescribeClassicalLBListeners | Query classic load balancer listener list | Yes | \ No newline at end of file diff --git a/translate/translated/04-best-practice/01-agent-advanced-config.md b/translate/translated/04-best-practice/01-agent-advanced-config.md index 34515f5c..778821e3 100644 --- a/translate/translated/04-best-practice/01-agent-advanced-config.md +++ b/translate/translated/04-best-practice/01-agent-advanced-config.md @@ -7,13 +7,15 @@ permalink: /best-practice/agent-advanced-config/ # Introduction -DeepFlow Agent Advanced Configuration. +DeepFlow Agent advanced configuration. -DeepFlow uses declarative APIs to control all deepflow-agents, with almost all deepflow-agent configurations being delivered through the deepflow-server. In DeepFlow, an agent-group is a group that manages the configuration of a set of deepflow-agents. We can specify the `vtap-group-id-request` in the local configuration file of the deepflow-agent (K8s ConfigMap, deepflow-agent.yaml on the Host) to declare the desired group to join, or directly configure the group each deepflow-agent belongs to on the deepflow-server (the latter has higher priority). The agent-group-config corresponds one-to-one with the agent-group and is associated through the agent-group ID. +DeepFlow uses a declarative API to centrally manage all agents, while the data collection configuration of agents is uniformly distributed by the deepflow-server to the agents within the corresponding agent-group based on the agent-group-config content. + +An agent-group is used to manage the configuration of a group of agents. By specifying `vtap-group-id-request` in the agent [configuration file](https://github.com/deepflowio/deepflow/blob/main/agent/config/deepflow-agent.yaml) (K8s ConfigMap or `/etc/deepflow-agent.yaml`), you can declare the agent-group it belongs to (if not specified, the [Default](../configuration/agent/) configuration is used by default). Finally, the association between agent, agent-group, and agent-group-config is established through the agent-group-id. ## Common Operations for agent-group -View the list of agent-groups: +View the agent-group list: ```bash deepflow-ctl agent-group list @@ -22,57 +24,56 @@ deepflow-ctl agent-group list Create an agent-group: ```bash -deepflow-ctl agent-group create your-agent-group +deepflow-ctl agent-group create ``` -Get the ID of the newly created agent-group: +View the created agent-group-id: ```bash -deepflow-ctl agent-group list your-agent-group +deepflow-ctl agent-group list ``` ## Common Operations for agent-group-config -Refer to the default agent configuration mentioned above, extract the parts you want to modify, create a `your-agent-group-config.yaml` file, and fill in the agent configuration parameters. Note that `vtap_group_id` must be included: +Refer to the [default configuration](../configuration/agent/) of agent-group-config, extract the parts that need to be modified, and output them to the `.yaml` file, for example: ```yaml -vtap_group_id: -# write configurations here +global: + limits: + max_millicpus: 2000 + max_memory: 4096 ``` -### Create agent-group-config +### Create an agent-group-config ```bash -deepflow-ctl agent-group-config create -f your-agent-group-config.yaml +deepflow-ctl agent-group-config create -f .yaml ``` -### Get the list of agent-group-config +### View the agent-group-config list ```bash deepflow-ctl agent-group-config list ``` -### Get the configuration of agent-group-config +### View the configuration of a specified agent-group-config ```bash -deepflow-ctl agent-group-config list -o yaml +deepflow-ctl agent-group-config list -o yaml ``` -### Get all configurations and their default values of agent-group-config +### View all default configurations of agent-group-config ```bash deepflow-ctl agent-group-config example ``` -### Update the configuration of agent-group-config +### Update the agent-group-config configuration ```bash -deepflow-ctl agent-group-config update -f your-agent-group-config.yaml +deepflow-ctl agent-group-config update -f .yaml ``` -## Common Configuration Items +## Description of Each Configuration Item -- `max_memory`: Maximum memory limit for the agent, default value is `768` MB. -- `thread_threshold`: Maximum number of threads for the agent, default value is `500`. -- `tap_interface_regex`: Regular expression configuration for the agent's collection network interface, default value is `^(tap.*|cali.*|veth.*|eth.*|en[ospx].*|lxc.*|lo)$`. The agent only needs to collect Pod network interfaces and Node/Host physical network interfaces. -- `platform_enabled`: Used when the agent reports resources, for the domain of `agent-sync`. Only one domain of `agent-sync` is allowed per DeepFlow platform. \ No newline at end of file +For details, please refer to the [Configuration Manual](../configuration/agent/), where each parameter is explained in detail with usage examples. \ No newline at end of file diff --git a/translate/translated/04-best-practice/03-special-environment-deployment.md b/translate/translated/04-best-practice/03-special-environment-deployment.md index efeab60e..80ef2259 100644 --- a/translate/translated/04-best-practice/03-special-environment-deployment.md +++ b/translate/translated/04-best-practice/03-special-environment-deployment.md @@ -1,5 +1,5 @@ --- -title: Special Environment Agent Deployment +title: Agent Deployment in Special Environments permalink: /best-practice/special-environment-deployment/ --- @@ -7,31 +7,31 @@ permalink: /best-practice/special-environment-deployment/ # Special K8s CNI -In common K8s environments, the DeepFlow Agent can collect full-stack observability signals, as shown in the top left of the figure below: +In common K8s environments, DeepFlow Agent can collect full-stack observability signals, as shown in the upper left corner of the figure below: - When two Pods on the same Node communicate, data can be collected from two locations: eBPF Syscall and cBPF Pod NIC. - When two Pods on different Nodes communicate, data can be collected from three locations: eBPF Syscall, cBPF Pod NIC, and cBPF Node NIC. -![Data collection capabilities under different K8s CNIs (Pod-to-Pod communication scenarios)](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20231114002715.png) +![Data collection capabilities under different K8s CNIs (Pod-to-Pod communication scenario)](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20231114002715.png) -However, under certain CNIs, due to the uniqueness of the traffic path, the data collected by the DeepFlow Agent will differ: +However, under certain CNIs, due to the particularity of the traffic path, the data collected by DeepFlow Agent will differ: -- In the Cilium CNI environment (top right of the figure above): - - Cilium [uses XDP](https://docs.cilium.io/en/stable/network/ebpf/intro/) to bypass the TCP/IP stack, resulting in only unidirectional traffic being visible on the Pod NIC named lxc-xxx. - - When two Pods on the same Node communicate, data can be collected from one location: eBPF Syscall. - - When two Pods on different Nodes communicate, data can be collected from two locations: eBPF Syscall and cBPF Node NIC, the latter collected from Node eth0. -- In MACVlan, [Huawei Cloud CCE Turbo](https://support.huaweicloud.com/usermanual-cce/cce_10_0284.html), and other CNI environments (bottom left of the figure above): - - Using MACVlan sub-interfaces instead of Veth-Pair + Bridge, there is no corresponding Pod NIC in the Root Netns, but all Pod traffic can be seen on Node eth0. - - In this case, the DeepFlow Agent can be configured with `tap_mode = 1 (virtual mirror)` to treat the traffic on the Node NIC as if it were collected on the Pod NIC. - - When two Pods on the same Node communicate, data can be collected from two locations: eBPF Syscall and cBPF Pod NIC, the latter collected from Node eth0. - - However, since there is only one copy of the communication traffic on eth0, the client and server share one set of cBPF Pod NIC data. - - When two Pods on different Nodes communicate, data can be collected from two locations: eBPF Syscall and cBPF Pod NIC, the latter collected from Node eth0. -- In the IPVlan CNI environment (bottom right of the figure above): - - Using IPVlan sub-interfaces instead of Veth-Pair + Bridge, there is no corresponding Pod NIC in the Root Netns, and only the traffic of Pods entering and leaving the Node can be seen on Node eth0. - - When two Pods on the same Node communicate, data can be collected from one location: eBPF Syscall. - - When two Pods on different Nodes communicate, data can be collected from two locations: eBPF Syscall and cBPF Node NIC, the latter collected from Node eth0. +- In a Cilium CNI environment (upper right corner of the figure above): + - Cilium [uses XDP](https://docs.cilium.io/en/stable/network/ebpf/intro/) to bypass the TCP/IP protocol stack, resulting in only unidirectional traffic being visible on the Pod NIC named lxc-xxx. + - When two Pods on the same Node communicate, data can be collected only from the eBPF Syscall location. + - When two Pods on different Nodes communicate, data can be collected from eBPF Syscall and cBPF Node NIC, the latter collected from Node eth0. +- In MACVlan, [Huawei Cloud CCE Turbo](https://support.huaweicloud.com/usermanual-cce/cce_10_0284.html), and other CNI environments (lower left corner of the figure above): + - MACVlan sub-interfaces are used instead of Veth-Pair + Bridge. In this case, there is no corresponding Pod NIC in the Root Netns, but all Pod traffic can be seen on Node eth0. + - In this case, DeepFlow Agent can be configured as described below with `tap_mode = 1 (virtual mirror)`, treating the traffic on the Node NIC as if it were collected on the Pod NIC. + - When two Pods on the same Node communicate, data can be collected from eBPF Syscall and cBPF Pod NIC, the latter collected from Node eth0. + - However, since there is only one copy of the communication traffic on eth0, the client and server share the same cBPF Pod NIC data. + - When two Pods on different Nodes communicate, data can be collected from eBPF Syscall and cBPF Pod NIC, the latter collected from Node eth0. +- In an IPVlan CNI environment (lower right corner of the figure above): + - IPVlan sub-interfaces are used instead of Veth-Pair + Bridge. In this case, there is no corresponding Pod NIC in the Root Netns, and only Pod traffic entering and leaving the Node can be seen on Node eth0. + - When two Pods on the same Node communicate, data can be collected only from the eBPF Syscall location. + - When two Pods on different Nodes communicate, data can be collected from eBPF Syscall and cBPF Node NIC, the latter collected from Node eth0. -Additionally, eBPF XDP can also be used in conjunction with IPVlan (e.g., [Alibaba Cloud's Terway CNI](https://developer.aliyun.com/article/1221415)), where the traffic collection capability is equivalent to Cilium or IPVlan. +In addition, eBPF XDP can also be used in combination with IPVlan (for example, [Alibaba Cloud's Terway CNI](https://developer.aliyun.com/article/1221415)), in which case the traffic collection capability is equivalent to Cilium or IPVlan. Some reference materials: @@ -41,7 +41,9 @@ Some reference materials: ## MACVlan -When K8s uses the macvlan CNI, only a single virtual network card shared by all PODs can be seen under rootns. In this case, additional configuration of the deepflow-agent is required: +### Collecting Only NIC Traffic in RootNS + +When K8s uses macvlan CNI, only a single virtual NIC shared by all Pods can be seen in rootns. In this case, additional configuration is required for deepflow-agent: 1. Create agent-group and agent-group-config: @@ -92,23 +94,23 @@ When K8s uses the macvlan CNI, only a single virtual network card shared by all deepflow-ctl agent-group-config create -f macvlan-agent-group-config.yaml ``` -5. Modify the agent-group of deepflow-agent: +5. Modify the deepflow-agent's agent-group: ```bash kubectl edit cm -n deepflow deepflow-agent ``` - Add the configuration: + Add configuration: ```yaml vtap-group-id-request: g-xxxxx ``` - Stop the deepflow-agent: + Stop deepflow-agent: ```bash kubectl -n deepflow patch daemonset deepflow-agent -p '{"spec": {"template": {"spec": {"nodeSelector": {"non-existing": "true"}}}}}' ``` - Delete the macvlan agent through deepflow-ctl: + Delete the macvlan agent via deepflow-ctl: ```bash deepflow-ctl agent delete ``` - Start the deepflow-agent: + Start deepflow-agent: ```bash kubectl -n deepflow patch daemonset deepflow-agent --type json -p='[{"op": "remove", "path": "/spec/template/spec/nodeSelector/non-existing"}]' ``` @@ -117,36 +119,46 @@ When K8s uses the macvlan CNI, only a single virtual network card shared by all deepflow-ctl agent list ``` +### Collecting NIC Traffic in Both RootNS and PodNS + +Refer to [the documentation](../configuration/agent/#inputs.cbpf.af_packet.inner_interface_capture_enabled) to enable `inputs.cbpf.af_packet.inner_interface_capture_enabled` in deepflow-agent to collect NIC traffic in PodNS. + +Note that the following configurations also need to be adjusted: + +- `inputs.cbpf.af_packet.tunning.ring_blocks_enabled`: Allows AF_PACKET memory consumption to be adjustable. +- `inputs.cbpf.af_packet.tunning.ring_blocks`: Allows reducing the total memory consumption of all AF_PACKETs. +- `inputs.cbpf.af_packet.inner_interface_regex`: Ensures correct matching of NIC names inside PodNS. + ## Huawei Cloud CCE Turbo Refer to the MACVlan configuration method. ## IPVlan -The only thing to note is that the tap_interface_regex of the collector only needs to be configured as the Node NIC list. +The only thing to note is that the Agent's tap_interface_regex only needs to be configured as the Node NIC list. ## Cilium eBPF -The only thing to note is that the tap_interface_regex of the collector only needs to be configured as the Node NIC list. +The only thing to note is that the Agent's tap_interface_regex only needs to be configured as the Node NIC list. -# Limited Permissions for Running Agent in K8s +# K8s Agent Permission Restrictions ## No Permission to Deploy K8s Daemonset -When there is no permission to run Daemonset in the Kubernetes cluster, but it is possible to run ordinary processes directly on the K8s Node, this method can be used to deploy the Agent. +When you do not have permission to run a Daemonset in a Kubernetes cluster but can run regular processes directly on K8s Nodes, you can use this method to deploy the Agent. -- Deploy a deepflow-agent in the form of a deployment - - By setting the environment variable `ONLY_WATCH_K8S_RESOURCE`, this agent only performs list-watch of K8s resources and sends them to the controller - - All other functions of this agent will be automatically disabled - - When the agent requests the server, it informs that it is watching K8s, and the server updates this information in the MySQL database - - This Agent, which is only used as a Watcher, will not appear in the Agent list -- In this K8s cluster, run a regular-function deepflow-agent on each K8s Node in the form of a Linux process - - Since these agents do not have the `IN_CONTAINER` environment variable, they will not list-watch K8s resources - - These agents will still obtain the IP and MAC addresses of the POD and synchronize them to the server - - These agents will complete all observability data collection functions - - The server will issue the Agent type as K8s to these agents +- Deploy a deepflow-agent as a deployment + - By setting the environment variable `K8S_WATCH_POLICY=watch-only`, this agent will only perform list-watch of K8s resources and send them to the controller. + - All other functions of this agent will be automatically disabled. + - When the agent requests the server, it informs that it is watch-k8s, and the server updates this information in the MySQL database. + - This Agent, used only as a Watcher, will not appear in the Agent list. +- In this K8s cluster, run a regular-function deepflow-agent as a Linux process on each K8s Node + - Since these agents do not have the `IN_CONTAINER` environment variable, they will not list-watch K8s resources. + - These agents will still obtain the IP and MAC addresses of Pods and synchronize them to the server. + - These agents will perform all observability data collection functions. + - The server will issue the Agent type as K8s to these agents. -### Deploy DeepFlow Agent in Deployment Mode +### Deploying DeepFlow Agent in Deployment Mode ```bash cat << EOF > values-custom.yaml @@ -159,26 +171,26 @@ helm install deepflow -n deepflow deepflow/deepflow-agent --create-namespace \ -f values-custom.yaml ``` -After deployment, a Domain (corresponding to this K8s cluster) will be automatically created. Obtain the `kubernetes-cluster-id` of the `your-cluster-name` cluster from `deepflow-ctl domain list`, and then continue with the following operations. +After deployment, a Domain (corresponding to this K8s cluster) will be automatically created. Obtain the `kubernetes-cluster-id` of the `your-cluster-name` cluster from `deepflow-ctl domain list`, and then proceed with the following steps. -### Deploy DeepFlow Agent in Ordinary Process Form +### Deploying DeepFlow Agent as a Regular Process -- Refer to [Deploy DeepFlow Agent on Traditional Servers](../ce-install/legacy-host/), but there is no need to create a Domain -- Modify the agent configuration file `/etc/deepflow-agent/deepflow-agent.yaml`, and fill in the cluster ID obtained in the previous step for `kubernetes-cluster-id` +- Refer to [Deploying DeepFlow Agent on Legacy Hosts](../ce-install/legacy-host/), but there is no need to create a Domain. +- Modify the agent configuration file `/etc/deepflow-agent/deepflow-agent.yaml`, and fill in the cluster ID obtained in the previous step for `kubernetes-cluster-id`. ## Daemonset Not Allowed to Request apiserver -By default, DeepFlow Agent runs in K8s as a Daemonset. However, in some cases, to protect the apiserver from overload, Daemonset is not allowed to request the apiserver. In this case, a similar method to the "No Permission to Deploy Daemonset" described in this document can be used for deployment: +By default, DeepFlow Agent runs as a Daemonset in K8s. However, in some cases, to protect the apiserver from overload, the Daemonset is not allowed to request the apiserver. In this case, you can also use a similar method to "No Permission to Deploy Daemonset" described above: -- Deploy a deepflow-agent deployment, which is only responsible for list-watching the apiserver and synchronizing K8s resource information -- Deploy a deepflow-agent daemonset, where no Pod will list-watch the apiserver +- Deploy a deepflow-agent deployment that only performs list-watch of the apiserver and synchronizes K8s resource information. +- Deploy a deepflow-agent daemonset where no Pod will list-watch the apiserver. ## deepflow-agent Not Allowed to Request apiserver -deepflow-server relies on K8s resource information reported by deepflow-agent to achieve AutoTagging capability. When your environment does not allow deepflow-agent to directly watch the K8s apiserver, you can implement a pseudo-deepflow-agent specifically for synchronizing K8s resources. This pseudo-deepflow-agent needs to implement the following functions: +deepflow-server relies on K8s resource information reported by deepflow-agent to implement AutoTagging. When your environment does not allow deepflow-agent to directly watch the K8s apiserver, you can implement a dedicated pseudo-deepflow-agent to synchronize K8s resources. This pseudo-deepflow-agent needs to implement the following functions: -- Periodically list-watch the K8s apiserver to obtain the latest K8s resource information -- Report K8s resource information to deepflow-server via gRPC interface +- Periodically list-watch the K8s apiserver to obtain the latest K8s resource information. +- Call the deepflow-server gRPC interface to report K8s resource information. ### gRPC Interface @@ -188,39 +200,39 @@ The interface for reporting K8s resource information is ([GitHub code link](http rpc KubernetesAPISync(KubernetesAPISyncRequest) returns (KubernetesAPISyncResponse) {} ``` -The structure of the message reported by pseudo-deepflow-agent (GitHub code link is the same as above): +Structure of the message reported by pseudo-deepflow-agent (GitHub code link as above): ```protobuf message KubernetesAPISyncRequest { - // Unless otherwise specified, the following fields must be included. + // Unless otherwise specified, all fields below are required. // K8s cluster identifier. // Please use the value configured in deepflow-agent.yaml in the same K8s cluster. optional string cluster_id = 1; // Version number of the resource information. - // When K8s resource information changes, ensure that this version number also changes, usually using Linux Epoch. - // When K8s resource information does not change, carry the version number from the previous request, and entries do not need to be transmitted. - // Even if the resource information does not change, request deepflow-server periodically, ensuring the interval between two requests does not exceed 24 hours. + // When K8s resource information changes, ensure this version number also changes, usually using Linux Epoch. + // When K8s resource information has not changed, carry the previous version number, and entries do not need to be transmitted. + // Even if the resource information has not changed, periodically request deepflow-server, ensuring the interval between two requests does not exceed 24 hours. optional uint64 version = 2; - // Error information. - // When there is an error calling the K8s API or other errors occur, this field can be used to inform deepflow-server. - // When there is error_msg, it is recommended to carry the version and entries fields used in the previous request. + // Error message. + // When calling the K8s API fails or other errors occur, use this field to inform deepflow-server. + // When error_msg exists, it is recommended to carry the version and entries fields used in the previous request. optional string error_msg = 3; // Source IP address. - // Usually, it can be filled in as the client IP address used when pseudo-deepflow-agent requests gRPC. - // Using a representative and distinctive source_ip can facilitate checking deepflow-server logs to locate the requester. + // Usually, this can be the client IP address used by pseudo-deepflow-agent when making the gRPC request. + // Using a representative and distinctive source_ip makes it easier to check deepflow-server logs to identify the requester. optional string source_ip = 5; // All information of various K8s resources. - // Note: It needs to include all types of all resources. Resources not appearing will be considered deleted by deepflow-server. + // Please note that all types of all resources must be included; resources not present will be considered deleted by deepflow-server. repeated common.KubernetesAPIInfo entries = 10; } message KubernetesAPIInfo { - // K8s resource type, currently supported resource types are: + // K8s resource type. Currently supported resource types are: // - *v1.Node // - *v1.Namespace // - *v1.Deployment @@ -234,16 +246,16 @@ message KubernetesAPIInfo { optional string type = 1; // List of resources of this type. - // Note: Please use JSON serialization, then use zlib for compression, and transmit the compressed byte stream. + // Note: Please serialize using JSON, then compress with zlib, and transmit the compressed byte stream. optional bytes compressed_info = 3; } ``` -The structure of the response message from deepflow-server (GitHub code link is the same as above): +Structure of the message replied by deepflow-server (GitHub code link as above): ```protobuf message KubernetesAPISyncResponse { - // The version number of the resource information accepted by deepflow-server, usually equal to the version in the most recent request. + // Version number of the resource information accepted by deepflow-server, usually equal to the version in the most recent request. optional uint64 version = 1; } ``` @@ -255,20 +267,20 @@ Note that deepflow-server requires certain K8s resource types to be reported, in - `*v1.Node` - `*v1.Namespace` - `*v1.Pod` -- `*v1.Deployment`, `*v1.StatefulSet`, `*v1.DaemonSet`, `*v1.ReplicationController`, `*v1.ReplicaSet`: Report as needed based on the workload type of the Pod +- `*v1.Deployment`, `*v1.StatefulSet`, `*v1.DaemonSet`, `*v1.ReplicationController`, `*v1.ReplicaSet`: Report as needed based on the Pod's workload type. -Other resources can be omitted: +Other resources are optional: - `*v1.Service` - `*v1beta1.Ingress` -For the above resources, the information that pseudo-deepflow-agent needs to report can be referenced [here](../features/auto-tagging/meta-tags/#依赖的-k8s-api). +For the above resources, the information that pseudo-deepflow-agent needs to report can be found in [this documentation](../features/auto-tagging/meta-tags/#依赖的-k8s-api). -# Limited Permissions for Running Agent on Cloud Servers +# Cloud Server Agent Permission Restrictions ## Running deepflow-agent as a Non-root User -Suppose we want to run the Agent installed in /usr/sbin/deepflow-agent using a regular user deepflow. We must first grant the necessary permissions to deepflow using the root user: +Suppose we want to run the Agent installed at /usr/sbin/deepflow-agent as a regular user named deepflow. We must first grant the necessary permissions to deepflow as the root user: ```bash ## Use the root user to grant execution permissions to the deepflow-agent @@ -286,11 +298,28 @@ Run deepflow-agent as a non-root user, for example: systemctl start deepflow-agent ``` -If you want to uninstall deepflow-agent, remember to remove the corresponding permissions: +If you want to uninstall deepflow-agent, be sure to remove the corresponding permissions: ```bash ## Use the root user to revoke execution permissions from the deepflow-agent setcap -r /usr/sbin/deepflow-agent rmdir /sys/fs/cgroup/cpu/deepflow-agent rmdir /sys/fs/cgroup/memory/deepflow-agent +``` + +# Agent Accessing Server via LB + +## Network Between Agent and Server Cluster Restricted, Requires LB Connection + +When deepflow-agent needs to access deepflow-server via an external load balancer, configure two load balancer listeners on the LB to forward to the deepflow-server's NodePort Service (port 30033 for agent registration, port 30035 for agent data reporting), and add the LB's address and port in the [agent-group-config](./../configuration/agent/). The specific configuration is as follows: + +> Note: Unless in special cases such as network isolation, it is not recommended to use the LB access solution. By default, the agent will automatically select the server node based on the data volume to achieve dynamic load balancing; when accessing via LB, the data reporting path will be limited by the LB's load strategy, which may cause uneven server node load. + +```yaml +global: + communication: + ingester_port: $LB_PORT + proxy_controller_port: $LB_PORT + ingester_ip: $LB_IP + proxy_controller_ip: $LB_IP ``` \ No newline at end of file diff --git a/translate/translated/04-best-practice/04-reduce-storage-overhead.md b/translate/translated/04-best-practice/04-reduce-storage-overhead.md index 4a2c6520..434e531a 100644 --- a/translate/translated/04-best-practice/04-reduce-storage-overhead.md +++ b/translate/translated/04-best-practice/04-reduce-storage-overhead.md @@ -5,155 +5,156 @@ permalink: /best-practice/reduce-storage-overhead/ > This document was translated by ChatGPT -This article introduces how to configure DeepFlow to reduce ClickHouse storage overhead. +This article explains how to configure DeepFlow to reduce ClickHouse storage overhead. -# Overview of Configuration Items +# Overview of Configuration Options -Before diving into specific configuration items, let's first look at the main data types collected by DeepFlow Agent. From the perspective of databases and tables in ClickHouse, the data mainly includes the following categories: +Before diving into specific configuration items, let’s first look at the main types of data collected by the DeepFlow Agent. From the perspective of databases and tables in ClickHouse, the data mainly falls into the following categories: -- `flow_log.l4_flow_log`: Flow logs. TCP/UDP five-tuple flow logs calculated based on cBPF traffic data. Each flow log contains packet header five-tuple fields, label fields, and performance metrics, consuming about 150 bytes of storage space on average. -- `flow_log.l7_flow_log`: Call logs. Call logs of application protocols such as HTTP/gRPC/MySQL calculated based on cBPF traffic data and eBPF function call data. Each call log contains key request/response fields, label fields, and performance metrics, consuming about 70 bytes of storage space on average, mainly depending on the length of the header fields. For details on the header fields stored by various application protocols, see [documentation](../features/l7-protocols/overview/). -- `flow_metrics`: Metric data. Metric data aggregated from flow logs and call logs, with default aggregation generating metrics with 1m and 1s time precision. Since these metrics are aggregated, they are relatively small, generally consuming only about 1/10 of the storage space of flow logs or call logs. -- `event.perf_event`: Performance events. Currently mainly stores file read/write events of processes. Each event contains fields such as process name, file name, and read/write performance metrics, consuming about 80 bytes per event. -- `profile`: Continuous profiling. Stores function call stacks of processes with continuous profiling enabled. By default, only the deepflow-agent and deepflow-server processes enable eBPF On-CPU Profile. +- `flow_log.l4_flow_log`: Flow logs. TCP/UDP five-tuple flow logs calculated from cBPF traffic data. Each flow log contains packet header five-tuple fields, tag fields, and performance metrics, consuming about 150 bytes of storage on average. +- `flow_log.l7_flow_log`: Call logs. Application protocol call logs (HTTP/gRPC/MySQL, etc.) calculated from cBPF traffic data and eBPF function call data. Each call log contains key request/response fields, tag fields, and performance metrics, consuming about 70 bytes of storage on average, depending mainly on the length of header fields. For details on stored header fields for various application protocols, see the [documentation](../features/l7-protocols/overview/). +- `flow_metrics`: Metrics data. Metrics aggregated from flow logs and call logs, by default generated at 1m and 1s time resolutions. Since these metrics are aggregated, their volume is small, typically consuming only about 1/10 of the storage space of flow logs or call logs. +- `event.perf_event`: Performance events. Currently stores process file read/write events, with each event containing process name, file name, read/write performance metrics, etc., consuming about 80 bytes per event. +- `profile`: Continuous profiling. Stores function call stacks for processes with continuous profiling enabled. By default, only the deepflow-agent and deepflow-server processes have eBPF On-CPU Profile enabled. -DeepFlow offers a variety of configurations to reduce ClickHouse storage overhead, summarized in the diagram below. +DeepFlow offers a wide range of configurations to reduce ClickHouse storage overhead, summarized in the diagram below. -![Configuration items to reduce data volume](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20231227002415.png) +![Configuration options to reduce data volume](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20231227002415.png) -The configuration parameters in the diagram can be divided into four categories based on their purpose: +The configuration parameters in the diagram can be categorized by purpose into four types: -- Black: Used to set data retention duration. Different types of data usually require different retention durations. -- Green: Used to reduce data granularity. By modifying the configuration, you can adjust the granularity of various data types, directly reducing storage pressure. -- Blue: Used to disable unimportant data. By modifying the configuration, you can disable some features you don't care about or filter out some traffic you don't care about, thereby reducing storage pressure. -- Red: Used to set overload protection thresholds. To protect themselves from overload, the Agent and Server expose threshold configuration items for data collection and storage rates, preventing the collection and writing of excessive data. +- Black: Set data retention periods. Different types of data usually require different retention periods. +- Green: Reduce data granularity. Adjusting these settings can change the granularity of various data types, directly reducing storage pressure. +- Blue: Disable unneeded data. Adjusting these settings can disable features you don’t care about or filter out traffic you don’t care about, thereby reducing storage pressure. +- Red: Set overload protection thresholds. Agents and servers expose rate limit settings for data collection and storage to protect themselves from overload, preventing excessive data collection and writes. # Estimating the Effect Before Making Changes -Before adjusting the configuration, you should confirm the storage consumption of the corresponding data in ClickHouse and pre-evaluate the benefits of the configuration adjustments. The following ClickHouse commands can help you evaluate the storage overhead of specific data tables. +Before adjusting configurations, you should check the storage consumption of the relevant data in ClickHouse to estimate the potential benefits. The following ClickHouse commands can help you assess the storage overhead of specific tables. -View the number of rows and space consumption of all data tables to identify which tables occupy the most storage space: +View the row count and space usage of all tables to identify which ones consume the most storage: ```sql -WITH sum(bytes_on_disk) AS size SELECT database, table, formatReadableSize(sum(data_uncompressed_bytes)) AS "Uncompressed Total Size", formatReadableSize(sum(bytes_on_disk)) AS "Compressed Total Size", sum(rows) AS "Total Rows", sum(data_uncompressed_bytes)/sum(rows) AS "Average Row Length (Uncompressed)", sum(bytes_on_disk)/sum(rows) AS "Average Row Length (Compressed)" FROM system.parts GROUP BY database, table ORDER BY size DESC; +WITH sum(bytes_on_disk) AS size SELECT database, table, formatReadableSize(sum(data_uncompressed_bytes)) AS "压缩前总大小", formatReadableSize(sum(bytes_on_disk)) AS "压缩后总大小", sum(rows) AS "总行数", sum(data_uncompressed_bytes)/sum(rows) AS "压缩前平均每行长度", sum(bytes_on_disk)/sum(rows) AS "压缩后平均每行长度" FROM system.parts GROUP BY database, table ORDER BY size DESC; WITH sum(bytes_on_disk) AS size SELECT database, table, - formatReadableSize(sum(data_uncompressed_bytes)) AS `Uncompressed Total Size`, - formatReadableSize(sum(bytes_on_disk)) AS `Compressed Total Size`, - sum(rows) AS `Total Rows`, - sum(data_uncompressed_bytes) / sum(rows) AS `Average Row Length (Uncompressed)`, - sum(bytes_on_disk) / sum(rows) AS `Average Row Length (Compressed)` + formatReadableSize(sum(data_uncompressed_bytes)) AS `压缩前总大小`, + formatReadableSize(sum(bytes_on_disk)) AS `压缩后总大小`, + sum(rows) AS `总行数`, + sum(data_uncompressed_bytes) / sum(rows) AS `压缩前平均每行长度`, + sum(bytes_on_disk) / sum(rows) AS `压缩后平均每行长度` FROM system.parts GROUP BY - database, table + database, + table ORDER BY size DESC -┌─database────────┬─table────────────────────────────────────┬─Uncompressed Total Size─┬─Compressed Total Size─┬────Total Rows─┬─Average Row Length (Uncompressed)─┬──Average Row Length (Compressed)─┐ -│ flow_log │ l7_flow_log_local │ 16.58 GiB │ 3.15 GiB │ 47231817 │ 377.02479955408023 │ 71.69953315579623 │ -│ flow_log │ l4_flow_log_local │ 4.84 GiB │ 1.51 GiB │ 10487780 │ 495.46614078479905 │ 154.46138525026268 │ -│ profile │ in_process_local │ 2.24 GiB │ 349.19 MiB │ 10325238 │ 233.1437255974148 │ 35.46215912892274 │ -│ flow_metrics │ network_map.1s_local │ 1.95 GiB │ 261.58 MiB │ 5267001 │ 398.20784256543715 │ 52.07621016210174 │ -│ flow_metrics │ network.1s_local │ 1.01 GiB │ 159.70 MiB │ 3027870 │ 357.85661273436443 │ 55.30619544432225 │ -│ flow_metrics │ application.1s_local │ 932.36 MiB │ 129.11 MiB │ 8225431 │ 118.85691983799998 │ 16.459093876053426 │ -│ deepflow_system │ deepflow_system_local │ 1.55 GiB │ 117.64 MiB │ 18478240 │ 89.79716217561845 │ 6.675736920832287 │ -│ flow_metrics │ application_map.1s_local │ 691.41 MiB │ 65.07 MiB │ 4128212 │ 175.6193790435181 │ 16.52821560520632 │ -│ flow_metrics │ network_map.1m_local │ 331.22 MiB │ 57.66 MiB │ 818085 │ 424.5371263377277 │ 73.90728347298875 │ -│ flow_metrics │ application_map.1m_local │ 228.35 MiB │ 28.72 MiB │ 1435956 │ 166.75001462440352 │ 20.973729000052927 │ -│ flow_metrics │ network.1m_local │ 121.96 MiB │ 27.08 MiB │ 328060 │ 389.8104432116076 │ 86.55356946899957 │ -│ flow_metrics │ application.1m_local │ 97.90 MiB │ 16.84 MiB │ 870320 │ 117.95215897600882 │ 20.283444020590128 │ -│ event │ perf_event_local │ 2.84 MiB │ 1.24 MiB │ 16543 │ 180.1239194825606 │ 78.70126337423683 │ -└─────────────────┴──────────────────────────────────────────┴─────────────────────────┴───────────────────────┴───────────────┴───────────────────────────────────┴──────────────────────────────────┘ +┌─database────────┬─table────────────────────────────────────┬─压缩前总大小─┬─压缩后总大小─┬────总行数─┬─压缩前平均每行长度─┬──压缩后平均每行长度─┐ +│ flow_log │ l7_flow_log_local │ 16.58 GiB │ 3.15 GiB │ 47231817 │ 377.02479955408023 │ 71.69953315579623 │ +│ flow_log │ l4_flow_log_local │ 4.84 GiB │ 1.51 GiB │ 10487780 │ 495.46614078479905 │ 154.46138525026268 │ +│ profile │ in_process_local │ 2.24 GiB │ 349.19 MiB │ 10325238 │ 233.1437255974148 │ 35.46215912892274 │ +│ flow_metrics │ network_map.1s_local │ 1.95 GiB │ 261.58 MiB │ 5267001 │ 398.20784256543715 │ 52.07621016210174 │ +│ flow_metrics │ network.1s_local │ 1.01 GiB │ 159.70 MiB │ 3027870 │ 357.85661273436443 │ 55.30619544432225 │ +│ flow_metrics │ application.1s_local │ 932.36 MiB │ 129.11 MiB │ 8225431 │ 118.85691983799998 │ 16.459093876053426 │ +│ deepflow_system │ deepflow_system_local │ 1.55 GiB │ 117.64 MiB │ 18478240 │ 89.79716217561845 │ 6.675736920832287 │ +│ flow_metrics │ application_map.1s_local │ 691.41 MiB │ 65.07 MiB │ 4128212 │ 175.6193790435181 │ 16.52821560520632 │ +│ flow_metrics │ network_map.1m_local │ 331.22 MiB │ 57.66 MiB │ 818085 │ 424.5371263377277 │ 73.90728347298875 │ +│ flow_metrics │ application_map.1m_local │ 228.35 MiB │ 28.72 MiB │ 1435956 │ 166.75001462440352 │ 20.973729000052927 │ +│ flow_metrics │ network.1m_local │ 121.96 MiB │ 27.08 MiB │ 328060 │ 389.8104432116076 │ 86.55356946899957 │ +│ flow_metrics │ application.1m_local │ 97.90 MiB │ 16.84 MiB │ 870320 │ 117.95215897600882 │ 20.283444020590128 │ +│ event │ perf_event_local │ 2.84 MiB │ 1.24 MiB │ 16543 │ 180.1239194825606 │ 78.70126337423683 │ +└─────────────────┴──────────────────────────────────────────┴──────────────┴──────────────┴───────────┴────────────────────┴─────────────────────┘ ``` -Query the daily space consumption of a specific data table, such as the daily space consumption of `l7_flow_log`, to evaluate how to set the retention duration for that type of data: +Check the daily space usage of a specific table, e.g., `l7_flow_log`, to evaluate how to set its retention period: ```sql -WITH sum(bytes_on_disk) AS size SELECT SUBSTRING(partition, 1, 10) AS date, formatReadableSize(sum(data_uncompressed_bytes)) AS "Uncompressed Total Size", formatReadableSize(sum(bytes_on_disk)) AS "Compressed Total Size", sum(rows) AS "Total Rows", sum(data_uncompressed_bytes)/sum(rows) AS "Average Row Length (Uncompressed)", sum(bytes_on_disk)/sum(rows) AS "Average Row Length (Compressed)" FROM system.parts WHERE `table` = 'l7_flow_log_local' GROUP BY date ORDER BY date DESC; +WITH sum(bytes_on_disk) AS size SELECT SUBSTRING(partition, 1, 10) AS date, formatReadableSize(sum(data_uncompressed_bytes)) AS "压缩前总大小", formatReadableSize(sum(bytes_on_disk)) AS "压缩后总大小", sum(rows) AS "总行数", sum(data_uncompressed_bytes)/sum(rows) AS "压缩前平均每行长度", sum(bytes_on_disk)/sum(rows) AS "压缩后平均每行长度" FROM system.parts WHERE `table` = 'l7_flow_log_local' GROUP BY date ORDER BY date DESC; WITH sum(bytes_on_disk) AS size SELECT substring(partition, 1, 10) AS date, - formatReadableSize(sum(data_uncompressed_bytes)) AS `Uncompressed Total Size`, - formatReadableSize(sum(bytes_on_disk)) AS `Compressed Total Size`, - sum(rows) AS `Total Rows`, - sum(data_uncompressed_bytes) / sum(rows) AS `Average Row Length (Uncompressed)`, - sum(bytes_on_disk) / sum(rows) AS `Average Row Length (Compressed)` + formatReadableSize(sum(data_uncompressed_bytes)) AS `压缩前总大小`, + formatReadableSize(sum(bytes_on_disk)) AS `压缩后总大小`, + sum(rows) AS `总行数`, + sum(data_uncompressed_bytes) / sum(rows) AS `压缩前平均每行长度`, + sum(bytes_on_disk) / sum(rows) AS `压缩后平均每行长度` FROM system.parts WHERE table = 'l7_flow_log_local' GROUP BY date ORDER BY date DESC -┌─date───────┬─Uncompressed Total Size─┬─Compressed Total Size─┬───Total Rows─┬─Average Row Length (Uncompressed)─┬─Average Row Length (Compressed)─┐ -│ 2023-12-27 │ 3.50 GiB │ 681.83 MiB │ 10049374 │ 374.3181802169966 │ 71.14405494312382 │ -│ 2023-12-26 │ 5.13 GiB │ 1012.92 MiB │ 14397814 │ 382.63502848418517 │ 73.76948966002756 │ -│ 2023-12-25 │ 5.73 GiB │ 1.07 GiB │ 16340447 │ 376.4205308459432 │ 70.36901530294735 │ -│ 2023-12-24 │ 2.22 GiB │ 436.62 MiB │ 6432796 │ 370.22603095139345 │ 71.17070586413746 │ -└────────────┴─────────────────────────┴───────────────────────┴──────────────┴───────────────────────────────────┴──────────────────────────────────┘ +┌─date───────┬─压缩前总大小─┬─压缩后总大小─┬───总行数─┬─压缩前平均每行长度─┬─压缩后平均每行长度─┐ +│ 2023-12-27 │ 3.50 GiB │ 681.83 MiB │ 10049374 │ 374.3181802169966 │ 71.14405494312382 │ +│ 2023-12-26 │ 5.13 GiB │ 1012.92 MiB │ 14397814 │ 382.63502848418517 │ 73.76948966002756 │ +│ 2023-12-25 │ 5.73 GiB │ 1.07 GiB │ 16340447 │ 376.4205308459432 │ 70.36901530294735 │ +│ 2023-12-24 │ 2.22 GiB │ 436.62 MiB │ 6432796 │ 370.22603095139345 │ 71.17070586413746 │ +└────────────┴──────────────┴──────────────┴──────────┴────────────────────┴────────────────────┘ ``` -Distinguish space consumption based on the value of a specific field. For example, view the number of rows for different application protocols (`l7_protocol`) in the `l7_flow_log` table and the average length of the `request_resource` field to evaluate whether to disable parsing for a certain application protocol: +Analyze space usage by a specific field value. For example, check the row count for each application protocol (`l7_protocol`) in `l7_flow_log` and the average length of the `request_resource` field to decide whether to disable parsing for certain protocols: ```sql -SELECT dictGet(flow_tag.int_enum_map, 'name', ('l7_protocol', toUInt64(l7_protocol))) AS "Application Protocol", count(0) AS "Number of Rows", sum(length(request_resource))/count(l7_protocol) AS "Average request_resource Length", sum(length(request_resource))/sum(if(request_resource !='', 1, 0)) AS "Average Non-empty request_resource Length" FROM flow_log.l7_flow_log WHERE time>now()-86400 GROUP BY l7_protocol ORDER BY "Number of Rows" DESC; +SELECT dictGet(flow_tag.int_enum_map, 'name_zh', ('l7_protocol', toUInt64(l7_protocol))) AS "应用协议", count(0) AS "行数", sum(length(request_resource))/count(l7_protocol) AS "平均 request_resource 长度", sum(length(request_resource))/sum(if(request_resource !='', 1, 0)) AS "平均非空 request_resource 长度" FROM flow_log.l7_flow_log WHERE time>now()-86400 GROUP BY l7_protocol ORDER BY "行数" DESC; SELECT - dictGet(flow_tag.int_enum_map, 'name', ('l7_protocol', toUInt64(l7_protocol))) AS `Application Protocol`, - count(0) AS `Number of Rows`, - sum(length(request_resource)) / count(l7_protocol) AS `Average request_resource Length`, - sum(length(request_resource)) / sum(if(request_resource != '', 1, 0)) AS `Average Non-empty request_resource Length` + dictGet(flow_tag.int_enum_map, 'name_zh', ('l7_protocol', toUInt64(l7_protocol))) AS `应用协议`, + count(0) AS `行数`, + sum(length(request_resource)) / count(l7_protocol) AS `平均 request_resource 长度`, + sum(length(request_resource)) / sum(if(request_resource != '', 1, 0)) AS `平均非空 request_resource 长度` FROM flow_log.l7_flow_log WHERE time > (now() - 86400) GROUP BY l7_protocol -ORDER BY `Number of Rows` DESC - -┌─Application Protocol─┬─────Number of Rows─┬─Average request_resource Length─┬─Average Non-empty request_resource Length─┐ -│ HTTP │ 15307526 │ 50.94938901296003 │ 51.16722749422727 │ -│ MySQL │ 12762197 │ 67.32974729977919 │ 103.38747090527679 │ -│ DNS │ 9271394 │ 41.6790123470106 │ 41.6790123470106 │ -│ HTTP2 │ 4561075 │ 0.9742264707333249 │ 1.0020964209295697 │ -│ TLS │ 3422769 │ 14.613528403465148 │ 23.354457466682167 │ -│ Redis │ 2140668 │ 89.61926931219601 │ 92.19873624192489 │ -│ gRPC │ 957471 │ 13.709391720480307 │ 23.696586596959204 │ -│ N/A │ 79113 │ 0 │ nan │ -│ Custom │ 4802 │ 1 │ 1 │ -│ PostgreSQL │ 1125 │ 81.112 │ 81.112 │ -│ MongoDB │ 108 │ 1.3703703703703705 │ 2.3125 │ -└──────────────────────┴────────────────────┴─────────────────────────────────┴───────────────────────────────────────────┘ +ORDER BY `行数` DESC + +┌─应用协议───┬─────行数─┬─平均 request_resource 长度─┬─平均非空 request_resource 长度─┐ +│ HTTP │ 15307526 │ 50.94938901296003 │ 51.16722749422727 │ +│ MySQL │ 12762197 │ 67.32974729977919 │ 103.38747090527679 │ +│ DNS │ 9271394 │ 41.6790123470106 │ 41.6790123470106 │ +│ HTTP2 │ 4561075 │ 0.9742264707333249 │ 1.0020964209295697 │ +│ TLS │ 3422769 │ 14.613528403465148 │ 23.354457466682167 │ +│ Redis │ 2140668 │ 89.61926931219601 │ 92.19873624192489 │ +│ gRPC │ 957471 │ 13.709391720480307 │ 23.696586596959204 │ +│ N/A │ 79113 │ 0 │ nan │ +│ Custom │ 4802 │ 1 │ 1 │ +│ PostgreSQL │ 1125 │ 81.112 │ 81.112 │ +│ MongoDB │ 108 │ 1.3703703703703705 │ 2.3125 │ +└────────────┴──────────┴────────────────────────────┴────────────────────────────────┘ ``` -View the number of rows for different observation points (`observation_point`) in the `l7_flow_log` table to evaluate whether to disable call logs collected at certain observation points: +Check the row count for each observation point (`observation_point`) in `l7_flow_log` to decide whether to disable call log collection at certain points: ```sql -SELECT observation_point, dictGet(flow_tag.string_enum_map, 'name', ('observation_point', observation_point)) AS "Observation Point", count(0) AS "Number of Rows" FROM flow_log.l7_flow_log WHERE time>now()-86400 GROUP BY observation_point ORDER BY "Number of Rows" DESC; +SELECT observation_point, dictGet(flow_tag.string_enum_map, 'name_zh', ('observation_point', observation_point)) AS "观测点", count(0) AS "行数" FROM flow_log.l7_flow_log WHERE time>now()-86400 GROUP BY observation_point ORDER BY "行数" DESC; SELECT observation_point, - dictGet(flow_tag.string_enum_map, 'name', ('observation_point', observation_point)) AS `Observation Point`, - count(0) AS `Number of Rows` + dictGet(flow_tag.string_enum_map, 'name_zh', ('observation_point', observation_point)) AS `观测点`, + count(0) AS `行数` FROM flow_log.l7_flow_log WHERE time > (now() - 86400) GROUP BY observation_point -ORDER BY `Number of Rows` DESC - -┌─observation_point─┬─Observation Point─┬─────Number of Rows─┐ -│ c │ Client NIC │ 15034372 │ -│ s │ Server NIC │ 9343403 │ -│ c-p │ Client Process │ 9179406 │ -│ s-p │ Server Process │ 5723075 │ -│ rest │ Other NIC │ 3170534 │ -│ s-nd │ Server Node │ 2796958 │ -│ c-nd │ Client Node │ 2334083 │ -│ local │ Local NIC │ 1365403 │ -│ s-app │ Server App │ 89280 │ -│ app │ App │ 80079 │ -│ s-gw │ Gateway to Server │ 11 │ -└───────────────────┴───────────────────┴────────────────────┘ +ORDER BY `行数` DESC + +┌─observation_point─┬─观测点─────────┬─────行数─┐ +│ c │ 客户端网卡 │ 15034372 │ +│ s │ 服务端网卡 │ 9343403 │ +│ c-p │ 客户端进程 │ 9179406 │ +│ s-p │ 服务端进程 │ 5723075 │ +│ rest │ 其他网卡 │ 3170534 │ +│ s-nd │ 服务端容器节点 │ 2796958 │ +│ c-nd │ 客户端容器节点 │ 2334083 │ +│ local │ 本机网卡 │ 1365403 │ +│ s-app │ 服务端应用 │ 89280 │ +│ app │ 应用 │ 80079 │ +│ s-gw │ 网关到服务端 │ 11 │ +└───────────────────┴────────────────┴──────────┘ ``` -Distinguish space consumption based on the value of a specific field. For example, view the number of rows for different durations in the `event.perf_event` table: +Analyze space usage by a specific field value. For example, check the row count for different durations in `event.perf_event`: ```sql SELECT count(), min(duration) AS min_duration_us, max(duration) AS max_duration_us, toUInt64(duration/1000) AS ms FROM event.perf_event GROUP BY ms ORDER BY ms ASC @@ -179,75 +180,75 @@ ORDER BY ms ASC ... ``` -# Black: Setting Data Retention Duration +# Black: Setting Data Retention Periods -Deepflow-server provides the ability to set retention durations for all data tables. You can search for `-ttl-hour` configuration items in server.yaml to set the retention duration for specific data tables as needed. Typically, you can set a longer retention duration for metric data (`flow_metrics`, `prometheus`, etc.) and a shorter retention duration for log data (`flow_log`, etc.). As you can see, the retention duration can be set with an accuracy of hours. +deepflow-server allows you to set retention periods for all tables. Search for the `-ttl-hour` option in `server.yaml` to set the retention period for specific tables as needed. Typically, you can set longer retention for metrics data (`flow_metrics`, `prometheus`, etc.) and shorter retention for log data (`flow_log`, etc.). Retention can be set with hourly precision. -In the enterprise edition, you can directly set the data retention duration on the DeepFlow page. The community edition only supports setting the retention duration for tables that have not been created yet. Therefore, if you want to modify the retention duration of a table, you need to first delete the table in ClickHouse and then restart deepflow-server. +In the Enterprise Edition, you can set retention periods directly in the DeepFlow UI. The Community Edition only supports setting retention for tables that have not yet been created, so to change the retention for an existing table, you must delete it in ClickHouse and restart deepflow-server. # Green: Reducing Data Granularity -Let's first focus on the green configuration items. Adjusting these settings can help us trade-off data granularity to reduce storage pressure. +Let’s first look at the green configuration items, which help you trade off data granularity to reduce storage pressure. For flow logs `flow_log.l4_flow_log`: -- Setting `l4_log_ignore_tap_sides` can discard flow logs collected at certain observation points (`observation_point`). As shown in the figure below, in a K8s container environment, DeepFlow by default collects flow logs on both virtual and physical NICs. When Pod1 accesses Pod3, we will collect four flow logs for the same traffic on four NICs along the way. For example, we can set this configuration item to c-nd and s-nd to discard flow logs collected on physical NICs. For a detailed description of observation points, refer to the [documentation](../features/universal-map/auto-metrics/#观测点说明). -- Setting `l4_log_tap_types` can discard flow logs collected at certain `network positions`. When we let the Agent handle the mirrored traffic of physical switches (enterprise version feature), this configuration item can be set to control the dropping of flow logs from specified mirror locations. Specifically, setting this configuration item to `[-1]` will completely disable flow log data. +- `l4_log_ignore_tap_sides`: Discard flow logs collected at certain observation points (`observation_point`). In a K8s container environment, DeepFlow collects flow logs from both virtual and physical NICs by default. For example, when Pod1 accesses Pod3, the same traffic is logged four times along the path. Setting this to `c-nd` and `s-nd` discards logs from physical NICs. See [documentation](../features/universal-map/auto-metrics/#观测点说明) for details on observation points. +- `l4_log_tap_types`: Discard flow logs collected at certain `network locations`. When processing mirrored traffic from physical switches (Enterprise Edition), you can use this to drop logs from specific mirror locations. Setting to `[-1]` completely disables flow logs. -![Observation Points of Data](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20231226212513.png) +![Observation points of data (observation_point)](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20231226212513.png) For call logs `flow_log.l7_flow_log`: -- Setting `obfuscate-enabled-protocols` can desensitize the `request_resource` field in call logs by replacing variables with `?`. Currently, desensitization is supported for MySQL, PostgreSQL, and Redis protocols. Desensitized fields will have a significantly reduced length, thereby reducing storage costs. Note that desensitization will increase the CPU overhead of the deepflow-agent. -- Setting `l7_log_ignore_tap_sides` can discard call logs collected at certain observation points (`observation_point`). In a K8s container environment, DeepFlow by default collects call logs on application processes, virtual NICs, and physical NICs. As shown in the figure above, when Pod1 accesses Pod3, we will collect six call logs for the same traffic across two processes and four NICs along the way. For example, we can set this configuration item to c-nd and s-nd to discard call logs collected on physical NICs. -- Setting `l7_log_tap_types` can discard call logs collected at certain `network positions`. When we let the Agent handle the mirrored traffic of physical switches (enterprise version feature), this configuration item can be set to control the dropping of call logs from specified mirror locations. Specifically, setting this configuration item to `[-1]` will completely disable call log data, also disabling distributed tracing functionality. +- `obfuscate-enabled-protocols`: Mask the `request_resource` field in call logs by replacing variables with `?`. Currently supports MySQL, PostgreSQL, and Redis. Masking reduces field length and thus storage usage but increases deepflow-agent CPU usage. +- `l7_log_ignore_tap_sides`: Discard call logs collected at certain observation points. In K8s, DeepFlow collects call logs from application processes, virtual NICs, and physical NICs. Setting this to `c-nd` and `s-nd` discards logs from physical NICs. +- `l7_log_tap_types`: Discard call logs collected at certain `network locations`. Setting to `[-1]` completely disables call logs and also disables distributed tracing. -Adjusting the above configurations for flow logs and call logs will not affect the accuracy of metrics data `flow_metrics`. Although metrics data does not consume much storage space, we also expose some configuration items: +Adjusting these settings for flow and call logs does not affect the accuracy of `flow_metrics`. For metrics data, although storage usage is small, there are also relevant settings: -- Setting `http-endpoint-extraction` can control how the endpoint field for the HTTP protocol is generated. For RPC (e.g., gRPC/Dubbo, etc.) protocols, the endpoint field is clear and can usually be obtained directly from the header. However, determining which part of the HTTP URI can be used as the endpoint is often challenging. By default, we extract the first two segments of all URIs as the endpoint. But `2` might be too short for some APIs and too long for others. Adjusting this configuration item can prevent generating exploding endpoint field values and also ensure the generated endpoints provide enough information. -- Setting `inactive_server_port_enabled` can store inactive TCP/UDP port numbers as `server_port = 0`. When there is port scanning traffic in the network, enabling this configuration can effectively reduce the cardinality of metric data. ClickHouse is insensitive to the cardinality of metrics, but reducing the number of metric data can make queries faster. This configuration item is enabled by default. -- Setting `inactive_ip_enabled` can store inactive IP addresses as `0.0.0.0`. When there is IP address scanning traffic in the network, enabling this configuration can effectively reduce the cardinality of metric data. This configuration item is enabled by default. -- Setting `vtap_flow_1s_enabled` can disable 1s granularity aggregate data. If high-precision metric data is not required, this configuration can be disabled. This configuration item is enabled by default. +- `http-endpoint-extration`: Controls how HTTP endpoints are generated. For RPC protocols (e.g., gRPC/Dubbo), endpoints are clear from headers. For HTTP URIs, by default, the first two segments are used. Adjust this to avoid excessive endpoint values or insufficiently informative ones. +- `inactive_server_port_enabled`: Stores inactive TCP/UDP ports as `server_port = 0`. Useful for reducing metric cardinality when port scanning traffic exists. Enabled by default. +- `inactive_ip_enabled`: Stores inactive IP addresses as `0.0.0.0`. Useful for reducing metric cardinality when IP scanning traffic exists. Enabled by default. +- `vtap_flow_1s_enabled`: Disables 1s granularity aggregation when high-resolution metrics are not needed. Enabled by default. For continuous profiling `profile`: -- Setting `on-cpu-profile.frequency` can adjust the sampling frequency of the function call stack. The default is 99, which means sampling 99 times per second, approximately collecting the function call stack every 10ms. Lowering this value can reduce the data volume of the function call stack. -- Setting `on-cpu-profile.cpu` can set whether to differentiate the function call stacks on different CPU cores. The default is 0, which means not differentiating CPU cores, resulting in low storage overhead. When set to 1, function call stacks on different CPU cores will not be aggregated. +- `on-cpu-profile.frequency`: Adjusts function call stack sampling frequency. Default is 99 (about every 10ms). Lowering reduces data volume. +- `on-cpu-profile.cpu`: Controls whether to distinguish call stacks by CPU core. Default 0 (no distinction, lower storage). Setting to 1 prevents aggregation across cores. -# Blue: Disabling Uninterested Data +# Blue: Disabling Unneeded Data -Next, let's look at the blue configuration items. Adjusting these settings can help us disable uninterested features or filter out uninterested traffic, thereby reducing storage costs. +Next, the blue configuration items help disable unneeded features or filter out irrelevant traffic to reduce storage usage. For call logs `flow_log.l7_flow_log`: -- Setting `l7-protocol-enabled` can select the application protocols to be parsed. By default, all application protocols are enabled for parsing. If you are not interested in certain protocols like Kafka or Redis, you can remove them from the list. -- Setting `l7-protocol-ports` can set the list of ports to try parsing for specific application protocols. By default, DNS only parses traffic on ports 53 and 5353, TLS only parses traffic on port 443, and all other application protocols parse traffic on all ports 1-65535. This configuration item is very effective for protocols with indistinct characteristics. For example, if you know all the communication ports for the Redis protocol in your environment, setting this configuration item can avoid mis-parsing, thereby preventing the storage of dirty data generated by mis-parsing. +- `l7-protocol-enabled`: Select which application protocols to parse. By default, all are enabled. Remove protocols like Kafka or Redis if not needed. +- `l7-protocol-ports`: Set the port list for parsing specific protocols. For example, DNS defaults to ports 53 and 5353, TLS to 443, others to all ports. Restricting ports for ambiguous protocols like Redis can prevent misparsing and dirty data. For metrics data `flow_metrics`: -- Setting `l4_performance_enabled` can disable the calculation and storage of advanced network performance metrics. Advanced network performance metrics include latency, performance, and exception metrics in the `network_*` tables (i.e., all metrics other than throughput). For a detailed list of metrics, see the [documentation](../features/universal-map/metrics-and-operators/#网络性能指标). -- Setting `l7_metrics_enabled` can disable the calculation and storage of application performance metrics. Application performance metrics correspond to the `application_*` tables. +- `l4_performance_enabled`: Disables calculation/storage of advanced network performance metrics (latency, performance, exceptions) in `network_*` tables. See [documentation](../features/universal-map/metrics-and-operators/#网络性能指标) for details. +- `l7_metrics_enabled`: Disables calculation/storage of application performance metrics in `application_*` tables. For performance events `event.perf_event`: -- Setting `io-event-collect-mode` can adjust the range of collected file read/write events. Setting it to 0 completely disables collection, setting it to 1 collects only file read/write events occurring within the Request lifecycle, and setting it to 2 collects all file read/write events. The default value is 1, focusing on monitoring the impact of file read/write on Request performance. -- Setting `io-event-minimal-duration` can adjust the number of recorded file read/write events by setting a duration threshold. The default value is 1ms, recording only events with a read/write duration of 1ms or more. Increasing this value can reduce the number of events. +- `io-event-collect-mode`: Adjusts the scope of file read/write event collection. 0 disables, 1 collects only within a Request lifecycle, 2 collects all. Default is 1. +- `io-event-minimal-duration`: Sets a duration threshold for recording events. Default is 1ms. Increasing reduces event count. For continuous profiling `profile`: -- Setting `on-cpu-profile.disabled` can completely disable the continuous profiling feature. +- `on-cpu-profile.disabled`: Completely disables continuous profiling. -Additionally, we can reduce the data volume from the source with some configuration items: +Additionally, you can reduce data at the source: -- Setting `tap_interface_regex` controls the list of NICs to collect traffic from. If we do not want to collect traffic from certain virtual or physical NICs, this configuration can be adjusted. -- Setting `capture_bpf` can use BPF expressions to filter collected traffic. For example, if we are sure that all traffic on port 9999 is monitoring data transmission, and we are not interested in it, we can set this item to `not port 9999` to filter out the collection of this traffic. -- Setting `kprobe_blacklist` can set a port number blacklist, allowing eBPF kprobe to use this port number list to filter (discard) Socket data, reducing the data volume of call logs. +- `tap_interface_regex`: Controls which NICs to capture traffic from. Adjust to exclude certain virtual or physical NICs. +- `capture_bpf`: Uses BPF expressions to filter captured traffic. For example, set to `not port 9999` to exclude monitoring traffic on port 9999. +- `kprobe_blacklist`: Sets a port blacklist for eBPF kprobe to filter (drop) socket data, reducing call log volume. # Red: Setting Overload Protection Thresholds -Finally, let's look at the red configuration items. We do not want the agent to output excessive flow logs and call logs in extreme situations, avoiding excessive bandwidth usage; nor do we want a single server's write volume to be too large, avoiding overloading ClickHouse. Therefore, we expose the following configuration items: +Finally, the red configuration items. We want to avoid agents outputting excessive flow or call logs in extreme cases to prevent high bandwidth usage, and avoid excessive write loads on a single server to prevent ClickHouse overload. The following settings are available: -- `l4_log_collect_nps_threshold` controls the maximum rate at which the agent sends flow logs. When the actual flow logs to be sent exceed this rate, the agent will actively discard them. This discarding behavior can be monitored through the `drop-in-throttle` metric in `deepflow_system.deepflow_agent_flow_aggr`. -- `l7_log_collect_nps_threshold` controls the maximum rate at which the agent sends call logs. When the actual call logs to be sent exceed this rate, the agent will actively discard them. This discarding behavior can be monitored through the `throttle-drop` metric in `deepflow_system.deepflow_agent_l7_session_aggr`. -- `l4-throttle` controls the maximum rate at which a single server replica writes flow logs. When set to 0, the value of the `throttle` configuration item is used. When the actual flow logs to be written exceed this rate, the server will actively discard them. This discarding behavior can be monitored through the `drop_count` metric in `deepflow_system.deepflow_server_ingester_decoder` (filtered using `msg_type: l4_log`). -- `l7-throttle` controls the maximum rate at which a single server replica writes call logs. When set to 0, the value of the `throttle` configuration item is used. When the actual call logs to be written exceed this rate, the server will actively discard them. This discarding behavior can be monitored through the `drop_count` metric in `deepflow_system.deepflow_server_ingester_decoder` (filtered using `msg_type: l7_log`). +- `l4_log_collect_nps_threshold`: Controls the maximum rate at which an agent sends flow logs. When exceeded, the agent drops logs. Monitor `drop-in-throttle` in `deepflow_system.deepflow_agent_flow_aggr` to track drops. +- `l7_log_collect_nps_threshold`: Controls the maximum rate at which an agent sends call logs. When exceeded, the agent drops logs. Monitor `throttle-drop` in `deepflow_system.deepflow_agent_l7_session_aggr` to track drops. +- `l4-throttle`: Controls the maximum write rate of flow logs per server replica. 0 uses the `throttle` value. When exceeded, the server drops logs. Monitor `drop_count` in `deepflow_system.deepflow_server_ingester_decoder` (filter `msg_type: l4_log`) to track drops. +- `l7-throttle`: Controls the maximum write rate of call logs per server replica. 0 uses the `throttle` value. When exceeded, the server drops logs. Monitor `drop_count` in `deepflow_system.deepflow_server_ingester_decoder` (filter `msg_type: l7_log`) to track drops. \ No newline at end of file diff --git a/translate/translated/04-best-practice/06-production-deployment.md b/translate/translated/04-best-practice/06-production-deployment.md index e9de3444..47d88862 100644 --- a/translate/translated/04-best-practice/06-production-deployment.md +++ b/translate/translated/04-best-practice/06-production-deployment.md @@ -9,11 +9,11 @@ permalink: /best-practice/production-deployment/ DeepFlow production deployment recommendations. -# Use LTS Version of DeepFlow +# Use the LTS Version of DeepFlow -Add the `--version 6.4.9` parameter to helm to install or upgrade the LTS version of DeepFlow Server and Agent. +Run `helm search repo deepflow -l` to check the [latest LTS version](../release-notes/release-timeline) -## Install LTS Version of DeepFlow Server +## Install the LTS Version of DeepFlow Server ::: code-tabs#shell @@ -23,7 +23,7 @@ Add the `--version 6.4.9` parameter to helm to install or upgrade the LTS versio # helm repo add deepflow https://deepflowio.github.io/deepflow helm repo update deepflow # use `helm repo update` when helm < 3.7.0 -helm upgrade --install deepflow -n deepflow deepflow/deepflow --version 6.4.9 --create-namespace +helm upgrade --install deepflow -n deepflow deepflow/deepflow --version 7.0.014 --create-namespace ``` @tab Use Aliyun @@ -40,13 +40,13 @@ helm repo update deepflow # use `helm repo update` when helm < 3.7.0 # image: # repository: registry.cn-beijing.aliyuncs.com/deepflow-ce/grafana # EOF -helm upgrade --install deepflow -n deepflow deepflow/deepflow --version 6.4.9 --create-namespace \ +helm upgrade --install deepflow -n deepflow deepflow/deepflow --version 7.0.014 --create-namespace \ -f values-custom.yaml ``` ::: -## Install LTS Version of DeepFlow Agent +## Install the LTS Version of DeepFlow Agent ### K8s Environment @@ -65,8 +65,7 @@ helm upgrade --install deepflow -n deepflow deepflow/deepflow --version 6.4.9 -- # helm repo add deepflow https://deepflowio.github.io/deepflow helm repo update deepflow # use `helm repo update` when helm < 3.7.0 -helm upgrade --install deepflow-agent -n deepflow deepflow/deepflow-agent --version 6.4.9 --create-namespace \ - -f values-custom.yaml +helm upgrade --install deepflow-agent -n deepflow deepflow/deepflow-agent --version 7.0.014 --create-namespace -f values-custom.yaml ``` @tab Use Aliyun @@ -84,8 +83,7 @@ helm upgrade --install deepflow-agent -n deepflow deepflow/deepflow-agent --vers # helm repo add deepflow https://deepflowio.github.io/deepflow helm repo update deepflow # use `helm repo update` when helm < 3.7.0 -helm upgrade --install deepflow-agent -n deepflow deepflow/deepflow-agent --version 6.4.9 --create-namespace \ - -f values-custom.yaml +helm upgrade --install deepflow-agent -n deepflow deepflow/deepflow-agent --version 7.0.014 --create-namespace -f values-custom.yaml ``` ::: @@ -99,7 +97,7 @@ Switch the Agent download link to the LTS version: @tab rpm ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/rpm/agent/v6.4.9/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-rpm.zip +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/rpm/agent/v7.0/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-rpm.zip unzip deepflow-agent-rpm.zip yum -y localinstall x86_64/deepflow-agent-1.0*.rpm ``` @@ -107,7 +105,7 @@ yum -y localinstall x86_64/deepflow-agent-1.0*.rpm @tab deb ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/deb/agent/v6.4.9/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-deb.zip +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/deb/agent/v7.0/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent-deb.zip unzip deepflow-agent-deb.zip dpkg -i x86_64/deepflow-agent-1.0*.systemd.deb ``` @@ -115,7 +113,7 @@ dpkg -i x86_64/deepflow-agent-1.0*.systemd.deb @tab binary file ```bash -curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/agent/v6.4.9/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent.tar.gz +curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/agent/v7.0/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-agent.tar.gz tar -zxvf deepflow-agent.tar.gz -C /usr/sbin/ cat << EOF > /etc/systemd/system/deepflow-agent.service @@ -140,18 +138,19 @@ systemctl daemon-reload ::: -## Install LTS Version of Cli +## Install the LTS Version of Cli Switch the Cli download link to the LTS version: ```bash -curl -o /usr/bin/deepflow-ctl https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/v6.4.9/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl +curl -o /usr/bin/deepflow-ctl https://deepflow-ce.oss-cn-beijing.aliyuncs.com/bin/ctl/v7.0/linux/$(arch | sed 's|x86_64|amd64|' | sed 's|aarch64|arm64|')/deepflow-ctl chmod a+x /usr/bin/deepflow-ctl ``` # Use Managed MySQL -In a production environment, it is recommended to use a managed MySQL to ensure availability. It is recommended to use MySQL version 8.0 or above. The following databases need to be created and authorized in advance: +In production environments, it is recommended to use managed MySQL to ensure availability. MySQL 8.0 or above is recommended. +You need to create the following databases in advance and grant account permissions: - deepflow - grafana @@ -172,7 +171,8 @@ mysql: # Use Managed ClickHouse -In a production environment, it is recommended to use a managed ClickHouse to ensure availability. It is recommended that the version of ClickHouse be at least 21.8. The following databases need to be created and authorized in advance: +In production environments, it is recommended to use managed ClickHouse to ensure availability. The recommended ClickHouse version is at least 21.8. +You need to create the following databases in advance and grant account permissions: - deepflow_system - event @@ -211,11 +211,11 @@ clickhouse: enabled: false ## Close ClickHouse deployment ``` -DeepFlow will write the IP:Port information of ClickHouse into an Endpoint of a Service. The controller and ingester of deepflow-server obtain the ClickHouse address list through the `list&watch` of this Service's Endpoint. The controller connects to all ClickHouse instances to create databases and table structures, while the ingester sorts all deepflow-server pod names and Endpoint IPs, mapping them to deepflow-server and ClickHouse in sequence, creating databases, table structures, and writing observability data. The querier accesses this Service to query observability data. +DeepFlow writes the IP:Port information of ClickHouse into a Service Endpoint. The controller and ingester of deepflow-server obtain the ClickHouse address list through `list&watch` of this Service Endpoint. The controller connects to all ClickHouse instances to create databases, table structures, etc. The ingester sorts all deepflow-server pod names and Endpoint IPs, maps them to deepflow-server and ClickHouse in order, and creates databases, table structures, and writes observability data. The querier queries observability data by accessing this Service. -Since ClickHouse needs to request MySQL, it is recommended to use managed MySQL along with managed ClickHouse. +Since ClickHouse needs to request MySQL, it is recommended to use managed MySQL together when using managed ClickHouse. -If only using managed ClickHouse without managed MySQL, it is recommended to open the NodePort of MySQL and configure `global.externalMySQL` to the NodePort access address. +If you only use managed ClickHouse without managed MySQL, it is recommended to open MySQL's NodePort and configure `global.externalMySQL` to the NodePort access address. `values-custom.yaml` configuration: @@ -256,28 +256,28 @@ mysql: type: NodePort ``` -If you want to reuse the port allocated by NodePort, you need to deploy twice. Before the second deployment, fill in the port allocated in the first deployment into `global.externalMySQL.port`. +If you want to reuse the port assigned by NodePort, you need to deploy twice, and before the second deployment, fill in the port assigned in the first deployment into `global.externalMySQL.port`. -Since ClickHouse will save the connection method of MySQL, after modifying the MySQL connection, you need to delete all databases in ClickHouse and restart deepflow-server to reset the database. +Since ClickHouse stores MySQL connection information, after modifying the MySQL connection, you need to delete all databases in ClickHouse and restart deepflow-server to reset the database. -# Optimize Traffic Path from deepflow-agent to deepflow-server +# Optimize the Traffic Path from deepflow-agent to deepflow-server -When deepflow-agent starts, it will use the `controller-ips` in the local configuration file (including ConfigMap) to request deepflow-server. deepflow-server will by default send the Node IP of the deepflow-server Pod to deepflow-agent (the Pod IP of deepflow-server is sent by default in the same cluster) for subsequent request configuration and data sending. When there are multiple deepflow-servers, different deepflow-server Node IPs will be sent for load balancing, and load balancing will be performed periodically. +When deepflow-agent starts, it uses the `controller-ips` in the local configuration file (including ConfigMap) to request deepflow-server. By default, deepflow-server sends the Node IP of the deepflow-server Pod to deepflow-agent (Pod IP in the same cluster) for subsequent configuration requests and data sending. When there are multiple deepflow-servers, different Node IPs are sent to deepflow-agent for load balancing, and re-sent periodically after load balancing. -At this time, two ports' IPs are dynamically sent by deepflow-server to deepflow-agent: +At this time, there are two ports whose IPs are dynamically sent from deepflow-server to deepflow-agent: -- deepflow-agent and deepflow-server are not in the same cluster +- deepflow-agent and deepflow-server are not in the same cluster: - Control plane 30035 - Data plane 30033 -- deepflow-agent and deepflow-server are in the same cluster - - Control plane 20035 (configured in deepflow-server ConfigMap as `controller.grpc-port`, default is 20035) - - Data plane 20033 (configured in deepflow-server ConfigMap as `ingester.listen-port`, default is 20033) +- deepflow-agent and deepflow-server are in the same cluster: + - Control plane 20035 (`controller.grpc-port` configured in deepflow-server ConfigMap, default 20035) + - Data plane 20033 (`ingester.listen-port` configured in deepflow-server ConfigMap, default 20033) -By default, deepflow-agent uses NodePort to connect to deepflow-server. This NodePort Service uses `externalTrafficPolicy=Cluster`, and the traffic from NodePort to deepflow-server will generally be forwarded again, occupying unnecessary inter-node bandwidth. In extreme cases, kube-proxy may occupy too much CPU and other resources due to excessive traffic. +By default, deepflow-agent connects to deepflow-server via NodePort. This NodePort Service uses `externalTrafficPolicy=Cluster`, and traffic from NodePort to deepflow-server is usually forwarded again, consuming unnecessary inter-node bandwidth. In extreme cases, kube-proxy may consume excessive CPU and other resources due to high traffic. -## Use LoadBalancer Type Service +## Use a LoadBalancer Type Service -In environments with LoadBalancer conditions, you can modify the Service type of deepflow-server to LoadBalancer, using LoadBalancer to proxy the traffic of deepflow-agent requests to deepflow-server, improving availability. +In environments with LoadBalancer capability, you can change the deepflow-server Service type to LoadBalancer to proxy deepflow-agent traffic to deepflow-server, improving availability. `values-custom.yaml` configuration: @@ -287,7 +287,7 @@ server: type: LoadBalancer ``` -After modifying the Service type of deepflow-server to LoadBalancer, you need to configure agent-group-config to switch the address of deepflow-server requested by deepflow-agent to the LoadBalancer IP: +After changing the deepflow-server Service type to LoadBalancer, you need to configure agent-group-config to switch the deepflow-agent request address to the LoadBalancer IP: ```yaml proxy_controller_ip: 1.2.3.4 # FIXME: Your LoadBalancer IP address @@ -296,11 +296,12 @@ proxy_controller_port: 30035 # The default is 30035 analyzer_port: 30033 # The default is 30033 ``` -Note: After configuration, this IP will be fixedly sent to the collector as the data transmission IP, and the collector will also fixedly use the `controller-ips` in the local configuration file to request the control plane port 30035 to obtain configuration information. +Note: After configuration, this IP will be fixed as the data transmission IP sent to the collector, and the collector will also always use the controller-ips in the local configuration file to request the control plane port 30035 for configuration information. ## Use Local externalTrafficPolicy -In environments without LoadBalancer conditions, you can configure the Service of deepflow-server to `externalTrafficPolicy=Local` to ensure that the traffic accessing the NodePort of a node will only be routed to the deepflow-server on that node. Due to the use of `externalTrafficPolicy=Local` and deepflow-server drift and other factors, some nodes' NodePorts may not be able to access deepflow-server. You need to be careful to avoid affecting the `controller-ip` in the configuration file of deepflow-agent. +In environments without LoadBalancer capability, you can configure the deepflow-server Service with `externalTrafficPolicy=Local` to ensure that traffic accessing a NodePort on a node is only routed to the deepflow-server on that node. +Due to `externalTrafficPolicy=Local` and deepflow-server migration, some NodePorts may not be able to access deepflow-server. Be careful to avoid affecting the controller-ip in the deepflow-agent configuration file. `values-custom.yaml` configuration: @@ -312,7 +313,7 @@ server: ## Use HostNetwork -Enable the HostNetwork of deepflow-server to reduce the pressure on kube-proxy. +Enable HostNetwork for deepflow-server to reduce kube-proxy load. `values-custom.yaml` configuration: @@ -322,31 +323,32 @@ server: dnsPolicy: ClusterFirstWithHostNet ``` -After enabling the HostNetwork of deepflow-server, you need to configure agent-group-config to switch the port requested by deepflow-agent to deepflow-server: +After enabling HostNetwork for deepflow-server, you need to configure agent-group-config to switch the ports used by deepflow-agent to request deepflow-server: ```yaml proxy_controller_port: 20035 # The deepflow-server controller listens on the port. The default port is 20035 analyzer_port: 20033 # The deepflow-server ingester listens on the port. The default port is 20033 ``` -# Integrate with Existing Grafana +# Integrate with an Existing Grafana -## Download and Install Plugins +## Download and Install the Plugin -DeepFlow supports integration with existing Grafana. It is recommended to use version 9.0 or above, with the minimum supported version being 8.0. Currently, DeepFlow's plugins are undergoing certification. Before the certification is completed, you need to configure Grafana to allow loading unsigned plugins: +DeepFlow supports integration with an existing Grafana. Grafana 9.0 or above is recommended, with a minimum supported version of 8.0. +Currently, DeepFlow's plugin is undergoing certification. Before certification is complete, you need to configure Grafana to allow loading unsigned plugins: ```ini [plugins] allow_loading_unsigned_plugins = deepflow-querier-datasource,deepflow-apptracing-panel,deepflow-topo-panel,deepflowio-tracing-panel,deepflowio-deepflow-datasource,deepflowio-topo-panel ``` -Download the plugin installation package: +Download the plugin package: ``` curl -O https://deepflow-ce.oss-cn-beijing.aliyuncs.com/pkg/grafana-plugin/stable/deepflow-gui-grafana.tar.gz ``` -Extract the downloaded plugin to the Grafana plugin directory, such as `/var/lib/grafana/plugins`, and restart Grafana to load the plugin: +Extract the downloaded plugin to the Grafana plugin directory, e.g., `/var/lib/grafana/plugins`, and restart Grafana to load the plugin: ```bash tar -zxvf deepflow-gui-grafana.tar.gz -C /var/lib/grafana/plugins/ @@ -356,15 +358,15 @@ tar -zxvf deepflow-gui-grafana.tar.gz -C /var/lib/grafana/plugins/ You can find DeepFlow Querier in Grafana Data sources and add the following configuration items: -- `Request Url`: The NodePort of the deepflow-server service querier port accessed by Grafana. Execute the following command to get the access address: +- `Request Url`: The NodePort of the deepflow-server service querier port accessed by Grafana. Run the following command to get the access address: ```bash echo "http://$(kubectl get nodes -o jsonpath="{.items[0].status.addresses[0].address}"):$(kubectl get --namespace deepflow -o jsonpath="{.spec.ports[0].nodePort}" services deepflow-server)" ``` -- `API Token`: No need to fill in +- `API Token`: Leave blank -- `Tracing Url`: The NodePort of the deepflow-app service app port accessed by Grafana. Execute the following command to open the NodePort and get the access address: +- `Tracing Url`: The NodePort of the deepflow-app service app port accessed by Grafana. Run the following command to open the NodePort and get the access address: `values-custom.yaml` configuration: ```yaml app: @@ -378,4 +380,4 @@ You can find DeepFlow Querier in Grafana Data sources and add the following conf ## Import Dashboard -Click to enter the newly added DeepFlow Data source, switch to the `Dashboards` page, and click `Import` on the dashboard to import the dashboard. +Click to enter the newly added DeepFlow Data source, switch to the `Dashboards` page, and click `Import` on the dashboard to import it. \ No newline at end of file diff --git a/translate/translated/04-best-practice/07-storage-engine-use-byconity.md b/translate/translated/04-best-practice/07-storage-engine-use-byconity.md new file mode 100644 index 00000000..8148f4e1 --- /dev/null +++ b/translate/translated/04-best-practice/07-storage-engine-use-byconity.md @@ -0,0 +1,374 @@ +--- +title: Using ByConity as the Storage Engine +permalink: /best-practice/storage-engine-use-byconity/ +--- + +> This document was translated by ChatGPT + +# Introduction + +[ByConity](https://byconity.github.io/docs/introduction/main-principle-concepts) is a project forked by ByteDance from ClickHouse (latest synced from ClickHouse v23.3), supporting compute-storage separation. + +Starting from version 6.6, DeepFlow supports choosing between ClickHouse and ByConity by adjusting deployment parameters. ClickHouse is used by default, but it can be switched to ByConity. + +::: tip +ByConity consists of 17 Pods, among which 9 Pods have a Request and Limit of 1.1C 1280M, 1 Pod has a Request and Limit of 1C 1G, and 1 Pod has a Request and Limit of 1C 512M. The local Disk Cache for components `byconity-server`, `vw-default`, and `vw-writer` can be modified via the `lru_max_size` configuration, and the log data storage limit can be modified via the `size` and `count` configurations. +Resource requirements: + +- CPU: It is recommended that the Kubernetes cluster has at least 12C allocatable resources available, though actual consumption will be higher. +- Memory: It is recommended that the Kubernetes cluster has at least 14G allocatable resources available, though actual consumption will be higher. +- Disk: It is recommended that each data node has more than 180G of disk capacity, with local Disk Cache for `byconity-server`, `vw-default`, and `vw-writer` each at 40G, and log data for `byconity-server`, `vw-default`, and `vw-writer` each at 20G. + ::: + +## Deployment Parameters + +ByConity connects to object storage by default, and environment requirements can be found in the official documentation. During deployment, simply add the byconity configuration to the custom `values-custom.yaml` file. +Note: In this example, Alibaba Cloud OSS is used. You must replace `endpoint`, `region`, `bucket`, `path`, `ak_id`, and `ak_secret` in the example with the correct parameters for your object storage. It is also recommended to adjust the replica count of `byconity-server`, `vw-default`, and `vw-writer` to match the number of `deepflow-server` instances or nodes. + +```yaml +global: + storageEngine: byconity + +clickhouse: + enabled: false + +byconity: + enabled: true + nameOverride: '' + fullnameOverride: '' + + image: + repository: '{{ .Values.global.image.repository }}/byconity' + tag: 1.0.0 + imagePullPolicy: IfNotPresent + fdbShell: + image: + repository: '{{ .Values.global.image.repository }}' + byconity: + configOverwrite: + storage_configuration: + cnch_default_policy: cnch_default_s3 + disks: + server_s3_disk_0: # FIXME + path: byconity0 + endpoint: https://oss-cn-beijing-internal.aliyuncs.com + region: cn-beijing + bucket: byconity + ak_id: XXXXXXX + ak_secret: XXXXXXX + type: bytes3 + is_virtual_hosted_style: true + + policies: + cnch_default_s3: + volumes: + bytes3: + default: server_s3_disk_0 + disk: server_s3_disk_0 + + ports: + tcp: 9000 + http: 8123 + rpc: 8124 + tcpSecure: 9100 + https: 9123 + exchange: 9410 + exchangeStatus: 9510 + + usersOverwrite: + users: + default: + password: '' + probe: + password: probe + profiles: + default: + allow_experimental_live_view: 1 + enable_multiple_tables_for_cnch_parts: 1 + + server: + replicas: 1 # FIXME + image: '' + podAnnotations: {} + resources: {} + hostNetwork: false + nodeSelector: {} + tolerations: [] + affinity: + nodeAffinity: {} + imagePullSecrets: [] + securityContext: {} + storage: + localDisk: + pvcSpec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 30Gi + storageClassName: openebs-hostpath # FIXME: replace to your storageClassName + log: + pvcSpec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 20Gi + storageClassName: openebs-hostpath # FIXME: replace to your storageClassName + configOverwrite: + logger: + level: trace + disk_cache_strategies: + simple: + lru_max_size: 429496729600 # 400Gi + # timezone: Etc/UTC + + tso: + replicas: 1 + image: '' + podAnnotations: {} + resources: {} + hostNetwork: false + nodeSelector: {} + tolerations: [] + affinity: {} + imagePullSecrets: [] + securityContext: {} + configOverwrite: {} + additionalVolumes: {} + storage: + localDisk: + pvcSpec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 10Gi + storageClassName: openebs-hostpath # FIXME: replace to your storageClassName + log: + pvcSpec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 10Gi + storageClassName: openebs-hostpath # FIXME: replace to your storageClassName + + daemonManager: + replicas: 1 # Please keep single instance now, daemon manager HA is WIP + image: '' + podAnnotations: {} + resources: {} + hostNetwork: false + nodeSelector: {} + tolerations: [] + affinity: {} + imagePullSecrets: [] + securityContext: {} + configOverwrite: {} + + resourceManager: + replicas: 1 + image: '' + podAnnotations: {} + resources: {} + hostNetwork: false + nodeSelector: {} + tolerations: [] + affinity: {} + imagePullSecrets: [] + securityContext: {} + configOverwrite: {} + + defaultWorker: &defaultWorker + replicas: 1 + image: '' + podAnnotations: {} + resources: {} + hostNetwork: false + nodeSelector: {} + tolerations: [] + affinity: {} + imagePullSecrets: [] + securityContext: {} + livenessProbe: + exec: + command: ['/opt/byconity/scripts/lifecycle/liveness'] + failureThreshold: 6 + initialDelaySeconds: 5 + periodSeconds: 10 + successThreshold: 1 + timeoutSeconds: 20 + readinessProbe: + exec: + command: ['/opt/byconity/scripts/lifecycle/readiness'] + failureThreshold: 5 + initialDelaySeconds: 10 + periodSeconds: 10 + successThreshold: 1 + timeoutSeconds: 10 + storage: + localDisk: + pvcSpec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 50Gi + storageClassName: openebs-hostpath #replace to your storageClassName + log: + pvcSpec: + accessModes: + - ReadWriteOnce + resources: + requests: + storage: 10Gi + storageClassName: openebs-hostpath #replace to your storageClassName + configOverwrite: + logger: + level: trace + disk_cache_strategies: + simple: + lru_max_size: 42949672960 # 40Gi + # timezone: Etc/UTC + + virtualWarehouses: + - <<: *defaultWorker + name: vw_default + replicas: 1 # FIXME + - <<: *defaultWorker + name: vw_write + replicas: 1 # FIXME + + commonEnvs: + - name: MY_POD_NAMESPACE + valueFrom: + fieldRef: + fieldPath: 'metadata.namespace' + - name: MY_POD_NAME + valueFrom: + fieldRef: + fieldPath: 'metadata.name' + - name: MY_UID + valueFrom: + fieldRef: + apiVersion: v1 + fieldPath: 'metadata.uid' + - name: MY_POD_IP + valueFrom: + fieldRef: + fieldPath: 'status.podIP' + - name: MY_HOST_IP + valueFrom: + fieldRef: + # fieldPath: "status.hostIP" + fieldPath: 'status.podIP' + - name: CONSUL_HTTP_HOST + valueFrom: + fieldRef: + fieldPath: 'status.hostIP' + + additionalEnvs: [] + + additionalVolumes: + volumes: [] + volumeMounts: [] + + postStart: '' + preStop: '' + livenessProbe: '' + readinessProbe: '' + + ingress: + enabled: false + + # For more detailed usage, please check fdb-kubernetes-operator API doc: https://github.com/FoundationDB/fdb-kubernetes-operator/blob/main/docs/cluster_spec.md + fdb: + enabled: true + enableCliPod: true + version: 7.1.15 + clusterSpec: + mainContainer: + imageConfigs: + - version: 7.1.15 + baseImage: '{{ .Values.global.image.repository }}/foundationdb' + tag: 7.1.15 + sidecarContainer: + imageConfigs: + - version: 7.1.15 + baseImage: '{{ .Values.global.image.repository }}/foundationdb-kubernetes-sidecar' + tag: 7.1.15-1 + processCounts: + stateless: 3 + log: 3 + storage: 3 + processes: + general: + volumeClaimTemplate: + spec: + storageClassName: openebs-hostpath #replace to your storageClassName + resources: + requests: + storage: 20Gi + + fdb-operator: + enabled: true + resources: + limits: + cpu: 1 + memory: 512Mi + requests: + cpu: 1 + memory: 512Mi + affinity: {} + image: + repository: '{{ .Values.global.image.repository }}/fdb-kubernetes-operator' + tag: v1.9.0 + pullPolicy: IfNotPresent + initContainerImage: + repository: '{{ $.Values.global.image.repository }}/foundationdb-kubernetes-sidecar' + initContainers: + 6.2: + image: + repository: '{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar' + tag: 6.2.30-1 + pullPolicy: IfNotPresent + 6.3: + image: + repository: '{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar' + tag: 6.3.23-1 + pullPolicy: IfNotPresent + 7.1: + image: + repository: '{{ $.Values.global.image.repository }}/foundationdb/foundationdb-kubernetes-sidecar' + tag: 7.1.15-1 + pullPolicy: IfNotPresent + hdfs: + enabled: false +``` + +Redeploy DeepFlow: + +```bash +helm del deepflow -n deepflow +helm install deepflow -n deepflow -f values-custom.yaml deepflow/deepflow +``` + +## Notes + +- ByConity only supports the AMD64 architecture. +- If some `byconity-fdb-storage` Pods fail to start, adjust the kernel parameters: + ```bash + sudo sysctl -w fs.inotify.max_user_watches=2099999999 + sudo sysctl -w fs.inotify.max_user_instances=2099999999 + sudo sysctl -w fs.inotify.max_queued_events=2099999999 + ``` +- ByConity depends on a FoundationDB cluster (FDB for short), which is used to store ByConity metadata. Deleting or rebuilding the FDB cluster will result in the loss of FDB data, which in turn will cause the loss of ByConity data. Therefore, the FDB component will not be deleted during the uninstallation of ByConity. If you do need to delete this component, execute the following command: + ```bash + kubectl delete FoundationDBCluster --all -n deepflow + ``` +- If using a private registry causes some FDB components to fail to pull images, you can resolve it with the following commands: + ```bash + kubectl patch serviceaccount default -p '{"imagePullSecrets": [{"name": "myregistrykey"}]}' -n deepflow + kubectl delete pod -n deepflow -l foundationdb.org/fdb-cluster-name=deepflow-byconity-fdb + ``` \ No newline at end of file diff --git a/translate/translated/04-best-practice/07-trouble-shooting-flow.md b/translate/translated/04-best-practice/07-trouble-shooting-flow.md deleted file mode 100644 index 33075823..00000000 --- a/translate/translated/04-best-practice/07-trouble-shooting-flow.md +++ /dev/null @@ -1,274 +0,0 @@ ---- -title: General Methods for Business Fault Diagnosis and Localization -permalink: /best-practice/trouble-shooting-flow/ ---- - -> This document was translated by ChatGPT - -# Overview - -## Unified Observability Data Lake - -The DeepFlow observability platform aggregates a vast array of observability data, including metrics, tracing, logging, profiling, and events, through eBPF collection and open data interfaces. - -With AutoTagging label injection technology, DeepFlow can inject all observability data with rich textual tags, including resource tags and business tags. These tags contain semantic information such as cloud resource information of application instances, container resource information, developers, maintainers, version numbers, commit_id, repository addresses marked in CI/CD pipelines, etc. By injecting tag information, during fault diagnosis, all metrics, tracing, logging, profiling, and events data related to a specific business or application can be retrieved with a single text field search and presented on a dashboard. Engineers can then analyze the data in 3-5 steps to reach a fault diagnosis conclusion. - -![Unified Observability Data Lake](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3281c6db18.jpeg) - -## Unified Collaboration Across Multiple Teams - -In common IT business system operations, business deployments span multiple availability zones, with complex architectures and numerous components. Operations and fault diagnosis involve extensive communication and collaboration between different teams, including applications, PaaS platforms, IaaS clouds, etc. The DeepFlow observability platform breaks down data silos, builds data associations, and provides unified observability and collaboration capabilities for operations teams across applications, PaaS, IaaS, and networks. - -![Unified Collaboration Across Multiple Teams](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080866b4b0683ed7b.jpeg) - -## From Macro to Micro, From One-Dimensional to Multi-Dimensional - -![Fault Diagnosis Process Starting from Applications](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080866b4b0696e08d.jpeg) - -In the DeepFlow observability platform, fault diagnosis generally follows a process of application RED metrics observation, call log retrieval, distributed tracing, multi-dimensional application diagnosis, multi-dimensional system diagnosis, and multi-dimensional network diagnosis. This process moves from macro to micro, from one-dimensional data observation to multi-dimensional data analysis, gradually answering the following five questions to diagnose the root cause of the problem: - -- **Who is in trouble?** -- **When it's in trouble?** -- **Which request is in trouble?** -- **Where is the root position?** -- **What is the root cause?** - -# Macro - Metrics Analysis and Early Warning - -## Application RED Metrics Observation - -In IT system operations, the **RED** metrics (Rate, Error, Duration) are commonly used as core monitoring indicators to evaluate the business quality/application service quality of the system: - -- **Rate** (Request Rate) - Represents the number of requests received per unit time, used to measure the throughput/pressure of the service. -- **Error** (Error Rate) - Represents the proportion of requests that return error responses, used to detect service anomalies. Errors are usually divided into client-side errors and server-side errors, with server-side errors being the primary focus of applications. -- **Duration** (Response Time) - Represents the time taken from request to response, used to detect slow service responses. Commonly observed statistics include "average response time," "P95 response time," and "P99 response time." - -In the DeepFlow platform, RED metrics for application calls are also used as the entry point for observability and fault diagnosis. Typically, filters are applied based on different dimensions such as namespace (pod_ns), container service (pod_service), workload (pod_group), and application call protocol (l7_protocol) to observe the RED metrics of the objects of interest. - -- Step 1: If you are an operations personnel for a specific application system, and the application modules are deployed and isolated in a K8s namespace named "A," you can use `pod_ns = A` to observe the RED metrics of all application services in that business system. -- Step 2: If you want to further narrow down the observation to a container service named "b" within the "A" namespace, you can add a `pod_svc = b` filter condition. -- Step 3: If you want to observe only the RED metrics for HTTP protocol calls, you can add a `l7_protocol = http` filter condition. - -At this point, you can start your observability journey with the DeepFlow platform. - -![DeepFlow Application RED Metrics Observation](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b32824d0259.jpeg) - -Of course, DeepFlow also provides more filter conditions for flexible use in different scenarios, which you can explore in future usage. - -By observing the RED metrics of application services, you can answer the **Who** and **When** questions, namely: - -- **Who is in trouble?** - Which observability object (e.g., a container service, a workload, a Pod, a specific path in the IT system) is experiencing service quality issues (response errors, slow responses, or timeouts) that require attention. -- **When it's in trouble?** - At what time did the anomaly occur? - -The next step is to quickly retrieve every abnormal application call at the abnormal time point and start micro-level observation of each abnormal application call. - -# Micro - Call Tracing and Triage - -## Call Log Retrieval - -The DeepFlow platform provides a hidden "right slide window" for each observability object. By clicking on any observability object in the metrics curve or metrics statistics list, the "right slide window" will automatically expand. The "right slide window" provides a series of data observation windows, including "application metrics," "endpoint list," "call logs," "network metrics," etc., for analyzing data from different dimensions of the observability object. - -The "call logs" in the "right slide window" can answer the **Which** question (**Which request is in trouble?**) - that is, which application call is abnormal? - -Retrieve all call logs at the abnormal time point and filter out the abnormal application calls (response errors, slow responses, or timeouts): - -![DeepFlow Call Log Retrieval](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3282b398ec.jpeg) - -## Distributed Tracing - -After identifying a single abnormal request, you can perform distributed tracing on the single abnormal request in the DeepFlow platform to answer the **Where** question (**Where is the root position of the trouble?**) - that is, find the root cause Span of response errors, slow responses, or timeouts through the distributed tracing flame graph. - -![DeepFlow Distributed Tracing Diagram](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3282d820d0.jpeg) - -Background Reading: - -- [3 Minutes to Understand DeepFlow Distributed Tracing Flame Graph](https://www.bilibili.com/video/BV1di421k7JE/) -- [3 Minutes to Understand the Implementation Principle of DeepFlow Distributed Tracing](https://www.bilibili.com/video/BV1ZC411E7ad/) - -## Common Examples and Analysis of Distributed Tracing Flame Graphs - -We use a simplified application service model to understand how to quickly find the **Root Position** through the DeepFlow distributed tracing flame graph. -In this scenario, the Client uses `http get` to access the front-end service, which queries the DNS service, accesses the MySQL database, makes an RPC call, and finally returns an `http response` to the Client. -![Simplified Application Service Model](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328390aa62.png) - -### Application Issues - -- Application Service - Slow IO Thread - -If there is a significant delay between the "POD NIC Span" and the "System Span" of the front-end service, it can be determined that the `http get` experienced a delay when entering the processing queue of the front-end service from the POD NIC queue. -Common causes include busy IO thread scheduling. - -![Flame Graph Example 1 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284a1f3bb.png) - -- Application Service - Slow Work Thread - -If there is a significant delay between the "System Span" of receiving the `http get` and the "System Span" of sending the `dns query` in the front-end service, it can be determined that the front-end service experienced a delay during internal processing. -Common causes include busy work thread scheduling. - -![Flame Graph Example 2 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284d7fe39.png) - -- Middleware - Slow DNS Service Response - -If the "System Span" of the DNS service shows a significant duration, it can be determined that the DNS service process took too long to query and return the DNS resolution result, directly causing the slow response of this business request. - -![Flame Graph Example 3 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328508566e.png) - -- Middleware - Slow MySQL Service Response - -Similar to the DNS service, if the "System Span" of the MySQL service shows a significant duration, it can be determined that the MySQL service process took too long to process and return the result, directly causing the slow response of this business request. - -![Flame Graph Example 4 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3285438f40.png) - -- Other Application Services - Slow RPC Service Response - -Similar to the DNS service, if the "System Span" of the RPC service shows a significant duration, it can be determined that the RPC service process took too long to process and return the result, directly causing the slow response of this business request. - -![Flame Graph Example 5 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328575653a.png) - -- Client - Slow Process Handling - -If the `http response` returned by the front-end service to the Client takes a while to reach the "System Span" after reaching the "POD NIC Span" of the Client, it can be determined that the `http response` experienced a delay when entering the processing queue of the Client process from the POD NIC queue. - -![Flame Graph Example 6 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3285982a40.png) - -### Network Issues - -- Network Transmission - Slow TCP Connection Establishment - -If there is a significant delay between the "System Span" and the "POD NIC Span" of the Client, it can be determined that the `http get` experienced a delay before entering the network. - -This situation generally occurs when the Client uses a short TCP connection, requiring the establishment of a TCP connection before sending the `http get`. Packet loss or delays during the TCP three-way handshake can cause this issue. - -![Flame Graph Example 7 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3283e248cf.png) - -- Network Transmission - Slow Transmission Within Client Container Node - -If there is a significant delay between the "POD NIC Span" and the "Node NIC Span" of the Client, it can be determined that the `http get` experienced a delay during transmission within the virtual network of the Client container node. - -![Flame Graph Example 8 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284069e45.png) - -- Network Transmission - Slow Transmission Between Container Nodes - -If there is a significant delay between the "Node NIC Span" of the Client and the "Node NIC Span" of the front-end service, it can be determined that the `http get` experienced a delay during transmission between the two container nodes. - -![Flame Graph Example 9 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284283c8e.png) - -- Network Transmission - Slow Transmission Within Server Container Node - -If there is a significant delay between the "Node NIC Span" and the "POD NIC Span" of the front-end service, it can be determined that the `http get` experienced a delay during transmission within the virtual network of the server container node. - -![Flame Graph Example 10 (Diagram)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328473bf8b.png) - -# Multi-Dimensional Analysis - Root Cause Diagnosis - -After the DeepFlow platform's application distributed tracing helps us answer the question "Where is the root position?", the next step is to conduct multi-dimensional data analysis around the **Root Position** to answer the question "What is the root cause?". - -- When distributed tracing determines the problem boundary is a specific application process, you can enter the "application diagnosis" phase to analyze and diagnose multiple dimensions of data for the application instance to determine the root cause of the application fault. -- When distributed tracing determines the problem boundary is due to network transmission, you can enter the "network diagnosis" phase to analyze and diagnose multiple dimensions of data for network transmission to determine the root cause of the network fault. -- When the problem is related to system performance (e.g., system CPU usage, system load, system interfaces), you can enter the "system diagnosis" phase to analyze and diagnose multiple dimensions of data for the operating system to determine the root cause of the operating system fault. - -## Application Diagnosis - -If the **Root Position** is a specific application instance, you can analyze the application instance's resource metrics (CPU, memory, disk, etc.), OnCPU continuous profiling, OffCPU continuous profiling, memory profiling, application metrics analysis, and application log retrieval in the DeepFlow platform to find the **Root Cause** within the application. - -### Application Instance Resource Metrics Analysis - -In the DeepFlow platform, you can integrate and uniformly observe and analyze the computing resource metrics of container Pods/Containers. By quickly identifying the metrics at the abnormal time point, you can determine whether container resources are the Root Cause. - -- Pod Status List Observation - -![POD Status Observation Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b32862b4f18.jpeg) - -- Container Metrics Details Observation - -![Container Metrics Observation Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328672066c.png) - -### OnCPU Continuous Profiling - -In the DeepFlow platform, you can perform continuous profiling of the application's OnCPU to identify CPU hotspot functions within the application process. - -![OnCPU Continuous Profiling Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3286e73055.png) - -### OffCPU Continuous Profiling - -In the DeepFlow platform, you can perform continuous profiling of the application's OffCPU to identify blocked functions within the application due to IO waits, locks, etc. - -![OffCPU Continuous Profiling Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3288a5dc39.png) - -### Memory Profiling - -In the DeepFlow platform, you can perform memory profiling of the application to identify memory hotspot functions within the application process. - -### Application Metrics Analysis - -In the DeepFlow platform, you can integrate and uniformly analyze the metrics actively exposed by the application to identify the Root Cause within the program through application metrics. - -### Application Log Analysis - -In the DeepFlow platform, you can integrate and uniformly analyze the logs actively printed by the application to identify the Root Cause within the program through application logs. - -## System Diagnosis - -### File IO Event Analysis - -When distributed tracing determines a specific "System Span" as the Root Position, you can immediately retrieve the list of slow file IO events accompanying that Span to determine whether file IO performance is the Root Cause. - -![File IO Event Analysis Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3287b2741a.png) - -### K8s Resource Change Event Analysis - -Analyze the list of K8s resource change events at the abnormal time point for the Root Position to determine whether container creation or destruction processes are the Root Cause. - -![K8s Resource Change Event Analysis Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3287e6aa0e.jpeg) - -### System Metrics Analysis - -In the DeepFlow platform, you can integrate and uniformly observe and analyze the system metrics of cloud servers and container nodes. By quickly identifying the system metrics at the abnormal time point, you can determine whether system resources are the Root Cause. - -- Host Metrics List Observation - -![Host Metrics List Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b32896e4f7a.png) - -### System Log Analysis - -In the DeepFlow platform, you can integrate and uniformly analyze the logs output by the system to identify the Root Cause within the system through system logs. - -## Network Diagnosis - -### Network Metrics Analysis - -When distributed tracing determines a specific "Network Span" as the Root Position, you can immediately retrieve the "network performance" associated with that application call to determine the key reasons for slow network transmission, such as: - -- TCP Connection Delay - The delay during the TCP three-way handshake process; -- TLS Connection Delay - The delay during the TLS connection process; -- Average Data Delay - The delay from request Data to response Data (average of multiple processes); -- Average System Delay - The delay from request Data to reply ACK message (average of multiple processes); -- Average Client Wait Delay - The delay from the last ACK message or response Data to the next request (average of multiple processes). - -![Network Metrics Analysis Example in Distributed Tracing](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328833d96a.png) - -### Flow Log Analysis - -When retrieving the "network performance" associated with the application call does not fully determine the Root Cause, you can further "view flow logs" to retrieve detailed data of the TCP session and determine whether there are retransmissions, zero windows, TCP RST, etc. - -![Key Information in Flow Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328a51deb5.jpeg) - -### TCP Sequence Analysis - -When flow logs alone do not provide a clear Root Cause, you can further investigate by viewing the corresponding "TCP Sequence Diagram" associated with the flow logs. This diagram allows you to examine the interaction process of each data packet within a TCP session, helping to identify anomalies in packet interaction sequences, unusual time gaps between packets, and other information that can lead to discovering the Root Cause. - -![TCP Sequence Diagram Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3289a9c9aa.png) - -- **TCP Sequence Diagram Root Cause Analysis Case Study** - - **Issue Description:** After the server responded to a client's data packet, no new business requests were received within 15 seconds, triggering a timeout and closing the TCP connection. However, 53 nanoseconds later, a new business request packet was received from the client. Since the TCP connection had already been closed on the server side, the server could not process the packet and had to send an RST message to notify the client to stop sending requests. - - **Impact:** The last application request went unanswered. - - **Solution:** Increase the TCP connection timeout duration on the server (longer than on the client side) and ensure that the client actively closes the TCP connection after each business transaction is completed. - -![TCP Sequence Diagram Issue Analysis Case Study](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328a3b7ad0.jpeg) - -### Network Device Metrics Analysis - -On the DeepFlow platform, you can also monitor the operational metrics of network devices through Telegraf integration. When distributed tracing identifies that the Root Cause is within the physical network, you can perform unified observation and analysis of the network device metrics to locate the Root Cause within the network infrastructure. diff --git a/translate/translated/05-features/01-l7-protocols/01-overview.md b/translate/translated/05-features/01-l7-protocols/01-overview.md index 17753578..f7a946f8 100644 --- a/translate/translated/05-features/01-l7-protocols/01-overview.md +++ b/translate/translated/05-features/01-l7-protocols/01-overview.md @@ -7,20 +7,25 @@ permalink: /features/l7-protocols/overview # Supported Application Protocols +To reduce resource overhead and avoid misidentification, the agent only parses the following application protocols by default: + +- HTTP, HTTP2/gRPC, MySQL, Redis, Kafka, DNS, TLS. + +To enable parsing of other application protocols, configure the agent's `l7-protocol-enabled`. All supported application protocols are listed as follows: [csv-L7 Protocol List](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/tag/enum/l7_protocol) -# Call Log Field Descriptions +# Call Log Field Description -The call log (`flow_log.l7_flow_log`) data table stores request logs of various protocols aggregated on a minute-by-minute basis, consisting of two main categories: Tag and Metrics fields. +The call log (`flow_log.l7_flow_log`) data table stores request logs for various protocols aggregated at a one-minute granularity, consisting of two main categories of fields: Tag and Metrics. ## Tags -Tag fields: These fields are primarily used for grouping and filtering. Detailed field descriptions are as follows. +Tag fields: Mainly used for grouping and filtering. Detailed field descriptions are as follows: -[csv-querier component database field descriptions](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/tag/flow_log/l7_flow_log.en) +[csv-Database field descriptions of the querier component](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/tag/flow_log/l7_flow_log.en) ## Metrics -Metrics fields: These fields are primarily used for calculations. Detailed field descriptions are as follows. +Metrics fields: Mainly used for calculations. Detailed field descriptions are as follows: -[csv-querier component database field descriptions](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_log/l7_flow_log.en) +[csv-Database field descriptions of the querier component](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_log/l7_flow_log.en) \ No newline at end of file diff --git a/translate/translated/05-features/01-l7-protocols/03-rpc.md b/translate/translated/05-features/01-l7-protocols/03-rpc.md index 366dd838..80f8aafc 100644 --- a/translate/translated/05-features/01-l7-protocols/03-rpc.md +++ b/translate/translated/05-features/01-l7-protocols/03-rpc.md @@ -7,43 +7,43 @@ permalink: /features/l7-protocols/rpc # Dubbo -Supports Hessian2 and Kryo serialization algorithms. By parsing the [Dubbo](https://dubbo.apache.org/en/docs3-v2/java-sdk/reference-manual/protocol/overview/) protocol, the fields of Dubbo Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: +Supports three serialization algorithms: Hessian2, Kryo, and Fastjson2. By parsing the [Dubbo](https://dubbo.apache.org/en/docs3-v2/java-sdk/reference-manual/protocol/overview/) protocol, the fields of Dubbo Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | --------------------- | ------------ | ------------------------ | ---------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | version | -- | -- | -| | request_type | 请求类型 | Method-Name | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | Service-Name | -- | -- | -| | request_id | 请求 ID | Request-ID | Request-ID | -- | -| | endpoint | 端点 | Service-Name/Method-Name | -- | -- | -| Resp. | response_code | 响应码 | -- | Status | -- | -| | response_status | 响应状态 | -- | Status | Normal: 20; Client Exception: 30/40/90; Server Exception: 31/50/60/70/80/100 | -| | response_exception | 响应异常 | -- | Status | Description of Status, refer to [Dubbo Protocol Details](https://dubbo.apache.org/zh/blog/2018/10/05/dubbo-%E5%8D%8F%E8%AE%AE%E8%AF%A6%E8%A7%A3/) | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | The `http_log_trace_id` configuration item of the Agent can define the name of the Header to be extracted | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | The `http_log_span_id` configuration item of the Agent can define the name of the Header to be extracted | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | attribute.rpc_service | -- | Service-Name | -- | -- | +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | -------------------- | ------------ | ------------------------ | ---------------- | -------------------------------------------------------------------------------------------------------------------------- | +| Req. | version | 协议版本 | version | -- | -- | +| | request_type | 请求类型 | Method-Name | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | Service-Name | -- | -- | +| | request_id | 请求 ID | Request-ID | Request-ID | -- | +| | endpoint | 端点 | Service-Name/Method-Name | -- | -- | +| Resp. | response_code | 响应码 | -- | Status | -- | +| | response_status | 响应状态 | -- | Status | Normal: 20; Client Exception: 30/40/90; Server Exception: 31/50/60/70/80/100 | +| | response_exception | 响应异常 | -- | Status | Description of Status, refer to [Dubbo Protocol Details](https://dubbo.apache.org/zh/blog/2018/10/05/dubbo-%E5%8D%8F%E8%AE%AE%E8%AF%A6%E8%A7%A3/) | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | The `http_log_trace_id` configuration item of the Agent can define the Header name to extract | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | The `http_log_span_id` configuration item of the Agent can define the Header name to extract | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | attribute.rpc_service| -- | Service-Name | -- | -- | **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ----------- | ----------- | ----------------------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | -- | Number of Responses | -| session_length | 会话长度 | -- | -- | Request Length + Response Length | -| request_length | 请求长度 | Data length | -- | -- | -| response_length | 响应长度 | -- | Data length | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client Exception + Server Exception | -| client_error | 客户端异常 | -- | Status | Refer to the description of Tag field `response_code` | -| server_error | 服务端异常 | -- | Status | Refer to the description of Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Exception / Response | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | +| Name | Chinese | Request | Response | Description | +| ------------------- | -------------- | ----------- | ----------- | --------------------------------------- | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | -- | -- | Number of Responses | +| session_length | 会话长度 | -- | -- | Request length + Response length | +| request_length | 请求长度 | Data length | -- | -- | +| response_length | 响应长度 | -- | Data length | -- | +| log_count | 日志总量 | -- | -- | -- | +| error | 异常 | -- | -- | Client Exception + Server Exception | +| client_error | 客户端异常 | -- | Status | Refer to the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | Status | Refer to the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Exception / Response | +| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | +| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | # gRPC @@ -51,43 +51,43 @@ By parsing the gRPC protocol, the fields of gRPC Request/Response are mapped to **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | --------------------- | ----------------- | ----------------- | ---------------- | ------------------------------------------------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | Version | -- | -- | -| | request_type | 请求类型 | Method | -- | -- | -| | request_domain | 请求域名 | Host or Authority | -- | -- | -| | request_resource | 请求资源 | Service-Name | -- | -- | -| | request_id | 请求 ID | Stream ID | Stream ID | -- | -| | endpoint | 端点 | Path | -- | -- | -| Resp. | response_code | 响应码 | -- | Status Code | -- | -| | response_status | 响应状态 | -- | Status Code | Normal: 1XX/2XX/3XX; Client Exception: 4XX; Server Exception: 5XX | -| | response_exception | 响应异常 | -- | Status Code | Description of Status Code, refer to [List of HTTP status codes](https://en.wikipedia.org/wiki/List_of_HTTP_status_codes) | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | The `http_log_trace_id` configuration item of the Agent can define the name of the Header to be extracted | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | The `http_log_span_id` configuration item of the Agent can define the name of the Header to be extracted | -| | x_request_id | X-Request-ID | X-Request-ID | X-Request-ID | The `http_log_x_request_id` configuration item of the Agent can define the name of the Header to be extracted | -| | http_proxy_client | HTTP Proxy Client | X-Forwarded-For | -- | The `http_log_proxy_client` configuration item of the Agent can define the name of the Header to be extracted | -| Misc. | attribute.rpc_service | -- | Service-Name | -- | -- | -| Misc. | attribute.x | -- | x | x | Supports collecting custom header fields [1] | - -- [1] The protocol header fields that need to be additionally collected can be defined through the static_config.l7-protocol-advanced-features.extra-log-fields in the collector configuration. For example, when User-Agent and Cookie are added in the configuration, the fields attribute.user_agent and attribute.cookie can be viewed in the call logs. +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | -------------------- | --------------- | ----------------- | ---------------- | ------------------------------------------------------------------------------------------------------------------- | +| Req. | version | 协议版本 | Version | -- | -- | +| | request_type | 请求类型 | Method | -- | -- | +| | request_domain | 请求域名 | Host or Authority | -- | -- | +| | request_resource | 请求资源 | Service-Name | -- | -- | +| | request_id | 请求 ID | Stream ID | Stream ID | -- | +| | endpoint | 端点 | Path | -- | -- | +| Resp. | response_code | 响应码 | -- | Status Code | -- | +| | response_status | 响应状态 | -- | Status Code | Normal: 1XX/2XX/3XX; Client Exception: 4XX; Server Exception: 5XX | +| | response_exception | 响应异常 | -- | Status Code | Description of Status Code, refer to [List of HTTP status codes](https://en.wikipedia.org/wiki/List_of_HTTP_status_codes) | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | The `http_log_trace_id` configuration item of the Agent can define the Header name to extract | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | The `http_log_span_id` configuration item of the Agent can define the Header name to extract | +| | x_request_id | X-Request-ID | X-Request-ID | X-Request-ID | The `http_log_x_request_id` configuration item of the Agent can define the Header name to extract | +| | http_proxy_client | HTTP Proxy Client | X-Forwarded-For | -- | The `http_log_proxy_client` configuration item of the Agent can define the Header name to extract | +| Misc. | attribute.rpc_service| -- | Service-Name | -- | -- | +| Misc. | attribute.x | -- | x | x | Supports collecting custom header fields [1] | + +- [1] The protocol header fields that need to be additionally collected can be defined through the static_config.l7-protocol-advanced-features.extra-log-fields in the collector configuration. For example, when User-Agent and Cookie are added to the configuration, the attribute.user_agent and attribute.cookie fields can be viewed in the call log. **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | HTTP2 Request Header | HTTP2 Response Header | Description | -| ------------------ | -------------- | -------------------- | --------------------- | ----------------------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | | -- | Number of Responses | -| session_length | 会话长度 | -- | -- | Request Length + Response Length | -| request_length | 请求长度 | Content-Length | -- | -- | -| request_length | 响应长度 | -- | Content-Length | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client Exception + Server Exception | -| client_error | 客户端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | -| server_error | 服务端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Exception / Response | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | +| Name | Chinese | HTTP2 Request Header | HTTP2 Response Header | Description | +| ------------------- | -------------- | -------------------- | --------------------- | --------------------------------------- | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | | -- | Number of Responses | +| session_length | 会话长度 | -- | -- | Request length + Response length | +| request_length | 请求长度 | Content-Length | -- | -- | +| request_length | 响应长度 | -- | Content-Length | -- | +| log_count | 日志总量 | -- | -- | -- | +| error | 异常 | -- | -- | Client Exception + Server Exception | +| client_error | 客户端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Exception / Response | +| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | +| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | # SOFARPC @@ -95,44 +95,44 @@ By parsing the [SOFARPC](https://blog.51cto.com/throwable/4896897) protocol, the **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ------------------------------- | --------------- | -------------------------------------------------------- | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_type | 请求类型 | method_name etc. [1] | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | target_service etc. [2] | -- | -- | -| | request_id | 请求 ID | req_id | req_id | -- | -| | endpoint | 端点 | $request_type/$request_resource | -- | -- | -| Resp. | response_code | 响应码 | -- | resp_code | -- | -| | response_status | 响应状态 | -- | resp_code | Normal: 0; Client Exception: 8; Server Exception: Others | -| | response_exception | 响应异常 | -- | -- | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | sofaTraceId etc. [3] | -- | -- | -| | span_id | SpanID | trace_context etc. [4] | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | - -- [1] sofa_head_method_name in the request header, or the methodName field of the com.alipay.sofa.rpc.core.request.SofaRequest class. -- [2] sofa_head_target_service in the request header, or the targetServiceUniqueName field of the com.alipay.sofa.rpc.core.request.SofaRequest class. -- [3] rpc_trace_context.sofaTraceId in the request header, or new_rpc_trace_context, or the sofaTraceId field of the com.alipay.sofa.rpc.core.request.SofaRequest class. -- [4] new_rpc_trace_context field in the request header. +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------ | ------------------------------- | --------------- | ----------------------------------------------- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_type | 请求类型 | method_name etc. [1] | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | target_service etc. [2] | -- | -- | +| | request_id | 请求 ID | req_id | req_id | -- | +| | endpoint | 端点 | $request_type/$request_resource | -- | -- | +| Resp. | response_code | 响应码 | -- | resp_code | -- | +| | response_status | 响应状态 | -- | resp_code | Normal: 0; Client Exception: 8; Server Exception: others | +| | response_exception | 响应异常 | -- | -- | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | sofaTraceId etc. [3] | -- | -- | +| | span_id | SpanID | trace_context etc. [4] | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +- [1] The sofa_head_method_name in the Request header, or the methodName field of the com.alipay.sofa.rpc.core.request.SofaRequest class. +- [2] The sofa_head_target_service in the Request header, or the targetServiceUniqueName field of the com.alipay.sofa.rpc.core.request.SofaRequest. +- [3] The rpc_trace_context.sofaTraceId in the Request header, or new_rpc_trace_context, or the sofaTraceId field of the com.alipay.sofa.rpc.core.request.SofaRequest class. +- [4] The new_rpc_trace_context field in the Request header. **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | ----------- | ----------------------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | -- | Number of Responses | -| session_length | 会话长度 | -- | -- | -- | -| request_length | 请求长度 | -- | -- | -- | -| request_length | 响应长度 | -- | -- | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client Exception + Server Exception | -| client_error | 客户端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | -| server_error | 服务端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Exception / Response | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | +| Name | Chinese | Request | Response | Description | +| ------------------- | -------------- | ------- | ----------- | --------------------------------------- | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | -- | -- | Number of Responses | +| session_length | 会话长度 | -- | -- | -- | +| request_length | 请求长度 | -- | -- | -- | +| request_length | 响应长度 | -- | -- | -- | +| log_count | 日志总量 | -- | -- | -- | +| error | 异常 | -- | -- | Client Exception + Server Exception | +| client_error | 客户端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Exception / Response | +| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | +| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | # FastCGI @@ -140,116 +140,116 @@ By parsing the [FastCGI](https://www.mit.edu/~yandros/doc/specs/fcgi-spec.html) **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ----------------------- | ---------------- | ------------------------------------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_type | 请求类型 | REQUEST_METHOD in PARAM | -- | -- | -| | request_domain | 请求域名 | HTTP_HOST in PARAM | -- | -- | -| | request_resource | 请求资源 | REQUEST_URI in PARAM | -- | -- | -| | request_id | 请求 ID | Request ID | Request ID | -- | -| | endpoint | 端点 | DOCUMENT_URI in PARAM | -- | -- | -| Resp. | response_code | 响应码 | -- | Status Code | Status in STDOUT, default 200 | -| | response_status | 响应状态 | -- | Status Code | Normal: 1XX/2XX/3XX; Client Exception: 4XX; Server Exception: 5XX | -| | response_exception | 响应异常 | -- | -- | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | The `http_log_trace_id` configuration item of the Agent can define the name of the Header to be extracted | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | The `http_log_span_id` configuration item of the Agent can define the name of the Header to be extracted | -| | x_request_id | X-Request-ID | X-Request-ID | X-Request-ID | The `http_log_x_request_id` configuration item of the Agent can define the name of the Header to be extracted | -| Misc. | -- | -- | -- | -- | -- | +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------ | ------------------------- | ---------------- | ---------------------------------------------------------------------- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_type | 请求类型 | REQUEST_METHOD in PARAM | -- | -- | +| | request_domain | 请求域名 | HTTP_HOST in PARAM | -- | -- | +| | request_resource | 请求资源 | REQUEST_URI in PARAM | -- | -- | +| | request_id | 请求 ID | Request ID | Request ID | -- | +| | endpoint | 端点 | DOCUMENT_URI in PARAM | -- | -- | +| Resp. | response_code | 响应码 | -- | Status Code | Status in STDOUT, default 200 | +| | response_status | 响应状态 | -- | Status Code | Normal: 1XX/2XX/3XX; Client Exception: 4XX; Server Exception: 5XX | +| | response_exception | 响应异常 | -- | -- | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | The `http_log_trace_id` configuration item of the Agent can define the Header name to extract | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | The `http_log_span_id` configuration item of the Agent can define the Header name to extract | +| | x_request_id | X-Request-ID | X-Request-ID | X-Request-ID | The `http_log_x_request_id` configuration item of the Agent can define the Header name to extract | +| Misc. | -- | -- | -- | -- | -- | **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | ----------- | ----------------------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | -- | Number of Responses | -| session_length | 会话长度 | -- | -- | -- | -| request_length | 请求长度 | -- | -- | -- | -| request_length | 响应长度 | -- | -- | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client Exception + Server Exception | -| client_error | 客户端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | -| server_error | 服务端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Exception / Response | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | +| Name | Chinese | Request | Response | Description | +| ------------------- | -------------- | ------- | ----------- | --------------------------------------- | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | -- | -- | Number of Responses | +| session_length | 会话长度 | -- | -- | -- | +| request_length | 请求长度 | -- | -- | -- | +| request_length | 响应长度 | -- | -- | -- | +| log_count | 日志总量 | -- | -- | -- | +| error | 异常 | -- | -- | Client Exception + Server Exception | +| client_error | 客户端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | Status Code | Refer to the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Exception / Response | +| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | +| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | # bRPC -By parsing the [bRPC](https://github.com/apache/brpc/blob/master/docs/cn/baidu_std.md) protocol, the fields of bRPC Request/Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationships are shown in the table below: - -**Tag Field Mapping Table - The following table only includes fields that have a mapping relationship** - -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ------------------------ | ------------------- | ----------------------------------------------- | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_type | 请求类型 | request.method_name | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | request.service_name | -- | -- | -| | request_id | 请求 ID | correlation_id | -- | correlation_id's high 32 bits of 64-bit integer | -| | endpoint | 端点 | service_name/method_name | -- | -- | -| Resp. | response_code | 响应码 | -- | Status Code | -- | -| | response_status | 响应状态 | -- | response.error_code | -- | -| | response_exception | 响应异常 | -- | response.error_text | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | request.log_id | -- | -- | -| Misc. | -- | -- | -- | -- | -- | - -**Metrics Field Mapping Table - The following table only includes fields that have a mapping relationship** - -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------- | ------------------------------------------------------ | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | -- | Number of Responses | -| session_length | 会话长度 | -- | -- | -- | -| request_length | 请求长度 | -- | -- | -- | -| response_length | 响应长度 | -- | -- | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client errors + Server errors | -| client_error | 客户端异常 | -- | -- | Refer to the `response_code` description in Tag fields | -| server_error | 服务端异常 | -- | -- | Refer to the `response_code` description in Tag fields | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | +By parsing the [bRPC](https://github.com/apache/brpc/blob/master/docs/cn/baidu_std.md) protocol, the fields of bRPC Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: + +**Tag Field Mapping Table, the following table only includes fields with mapping relationships** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------ | ------------------------ | ------------------- | ----------------------------------------- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_type | 请求类型 | request.method_name | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | request.service_name | -- | -- | +| | request_id | 请求 ID | correlation_id | -- | High 32 bits of the 64-bit correlation_id | +| | endpoint | 端点 | service_name/method_name | -- | -- | +| Resp. | response_code | 响应码 | -- | Status Code | -- | +| | response_status | 响应状态 | -- | response.error_code | -- | +| | response_exception | 响应异常 | -- | response.error_text | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | request.log_id | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** + +| Name | Chinese | Request | Response | Description | +| ------------------- | -------------- | ------- | -------- | ----------------------------------------- | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | -- | -- | Number of Responses | +| session_length | 会话长度 | -- | -- | -- | +| request_length | 请求长度 | -- | -- | -- | +| response_length | 响应长度 | -- | -- | -- | +| log_count | 日志总量 | -- | -- | -- | +| error | 异常 | -- | -- | Client Exception + Server Exception | +| client_error | 客户端异常 | -- | -- | Refer to the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | -- | Refer to the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Exception / Response | +| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | +| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | # Tars -By parsing the [Tars](https://doc.tarsyun.com/#/base/tars-protocol.md) protocol, the fields of Tars Request/Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationships are shown in the table below: - -**Tag Field Mapping Table - The following table only includes fields that have a mapping relationship** - -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ------------------------ | ------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------- | -| Req. | version | 协议版本 | tars_version | -- | -- | -| | request_type | 请求类型 | request.method_name | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | request.service_name | -- | -- | -| | request_id | 请求 ID | request_id | -- | -- | -| | endpoint | 端点 | service_name/method_name | -- | -- | -| Resp. | response_code | 响应码 | -- | Status Code | -- | -| | response_status | 响应状态 | -- | response.status | 0 for normal, -10 to -12 for client errors, others for server errors | -| | response_exception | 响应异常 | -- | Refer to the return code description, see [iRet Code](https://doc.tarsyun.com/#/base/tars-protocol.md) | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | - -**Metrics Field Mapping Table - The following table only includes fields that have a mapping relationship** - -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------- | ------------------------------------------------------ | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | -- | Number of Responses | -| session_length | 会话长度 | -- | -- | -- | -| request_length | 请求长度 | -- | -- | -- | -| response_length | 响应长度 | -- | -- | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client errors + Server errors | -| client_error | 客户端异常 | -- | -- | Refer to the `response_code` description in Tag fields | -| server_error | 服务端异常 | -- | -- | Refer to the `response_code` description in Tag fields | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | +By parsing the [Tars](https://doc.tarsyun.com/#/base/tars-protocol.md) protocol, the fields of Tars Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: + +**Tag Field Mapping Table, the following table only includes fields with mapping relationships** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------ | ------------------------ | --------------------------------------------------------------------------------- | ---------------------------------------------------- | +| Req. | version | 协议版本 | tars_version | -- | -- | +| | request_type | 请求类型 | request.method_name | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | request.service_name | -- | -- | +| | request_id | 请求 ID | request_id | -- | -- | +| | endpoint | 端点 | service_name/method_name | -- | -- | +| Resp. | response_code | 响应码 | -- | Status Code | -- | +| | response_status | 响应状态 | -- | response.status | 0 Normal, -10 to -12 Client Exception, others Server Exception | +| | response_exception | 响应异常 | -- | Refer to return code description, see [iRet Code](https://doc.tarsyun.com/#/base/tars-protocol.md) | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** + +| Name | Chinese | Request | Response | Description | +| ------------------- | -------------- | ------- | -------- | --------------------------------------- | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | -- | -- | Number of Responses | +| session_length | 会话长度 | -- | -- | -- | +| request_length | 请求长度 | -- | -- | -- | +| response_length | 响应长度 | -- | -- | -- | +| log_count | 日志总量 | -- | -- | -- | +| error | 异常 | -- | -- | Client Exception + Server Exception | +| client_error | 客户端异常 | -- | -- | Refer to the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | -- | Refer to the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Exception / Response | +| client_error_ratio | 客户端异常比例 | -- | -- | Client Exception / Response | +| server_error_ratio | 服务端异常比例 | -- | -- | Server Exception / Response | \ No newline at end of file diff --git a/translate/translated/05-features/01-l7-protocols/05-nosql.md b/translate/translated/05-features/01-l7-protocols/05-nosql.md index 4a63cf6a..1abacc2d 100644 --- a/translate/translated/05-features/01-l7-protocols/05-nosql.md +++ b/translate/translated/05-features/01-l7-protocols/05-nosql.md @@ -7,74 +7,110 @@ permalink: /features/l7-protocols/nosql # Redis -By parsing the [Redis](https://redis.io/docs/reference/protocol-spec/) protocol, the fields of Redis Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: - -**Tag Field Mapping Table, the following table only includes fields with mapping relationships** - -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ---------------- | ---------------- | ----------------------------------------- | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_type | 请求类型 | First word of Payload | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | Remaining characters of Payload | -- | -- | -| | request_id | 请求 ID | -- | -- | -- | -| | endpoint | 端点 | -- | -- | -- | -| Resp. | response_code | 响应码 | -- | -- | -- | -| | response_status | 响应状态 | -- | ERR message | Normal: No `ERR` message; Server error: ERR message | -| | response_exception | 响应异常 | -- | ERR message Payload | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | - -**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** - -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | --------------------------------------- | -------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| sql_affected_rows | SQL 影响行数 | -- | Affected Rows of `command complete` message | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client errors + Server errors | -| client_error | 客户端异常 | -- | -- | -- | -| server_error | 服务端异常 | -- | `ERR` message | Refer to the description of Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | +By parsing the [Redis](https://redis.io/docs/reference/protocol-spec/) protocol, the fields of Redis Request / Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationship is shown in the table below: + +**Tag field mapping table — only fields with mapping relationships are included below** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------- | ------------------ | ----------------- | --------------------------------------------------------------------------- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_type | 请求类型 | First word in Payload | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | Remaining characters in Payload | -- | -- | +| | request_id | 请求 ID | -- | -- | -- | +| | endpoint | 端点 | -- | -- | -- | +| Resp. | response_code | 响应码 | -- | -- | -- | +| | response_status | 响应状态 | -- | ERR message | Normal: no `ERR` message; Server error: ERR message | +| | response_exception | 响应异常 | -- | ERR message Payload | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +**Metrics field mapping table — only fields with mapping relationships are included below** + +| Name | Chinese | Request | Response | Description | +| ------------------- | --------------- | ------- | --------------------------------------- | ------------------------------------------------ | +| request | 请求 | -- | -- | Number of requests | +| response | 响应 | -- | | Number of responses | +| sql_affected_rows | SQL 影响行数 | -- | Affected Rows in `command complete` message | -- | +| log_count | 日志总量 | -- | -- | -- | +| error | 异常 | -- | -- | Client errors + Server errors | +| client_error | 客户端异常 | -- | -- | -- | +| server_error | 服务端异常 | -- | `ERR` message | See description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Errors / Responses | +| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | +| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | # MongoDB -By parsing the [MongoDB](https://www.mongodb.com/docs/manual/reference/mongodb-wire-protocol/) protocol, the fields of MongoDB Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: - -**Tag Field Mapping Table, the following table only includes fields with mapping relationships** - -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | -------------- | ---------------------- | ---------------- | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_type | 请求类型 | OpCode | -- | -- | -| | request_domain | 请求域名 | -- | -- | -- | -| | request_resource | 请求资源 | BodyDocument | -- | -- | -| | request_id | 请求 ID | -- | -- | -- | -| | endpoint | 端点 | -- | -- | -- | -| Resp. | response_code | 响应码 | -- | Code of BodyDocument | -- | -| | response_status | 响应状态 | -- | Code of BodyDocument | Judged by Code | -| | response_exception | 响应异常 | -- | errmsg of BodyDocument | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | - -**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** - -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | --------- | -------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| error | 异常 | -- | -- | Client errors + Server errors | -| client_error | 客户端异常 | -- | -- | -- | -| server_error | 服务端异常 | -- | `ERR` message | Refer to the description of Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | \ No newline at end of file +By parsing the [MongoDB](https://www.mongodb.com/docs/manual/reference/mongodb-wire-protocol/) protocol, the fields of MongoDB Request / Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationship is shown in the table below: + +**Tag field mapping table — only fields with mapping relationships are included below** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------- | -------------- | ------------------------ | ---------------------------- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_type | 请求类型 | OpCode | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | BodyDocument | -- | -- | +| | request_id | 请求 ID | -- | -- | -- | +| | endpoint | 端点 | -- | -- | -- | +| Resp. | response_code | 响应码 | -- | Code in BodyDocument | -- | +| | response_status | 响应状态 | -- | Code in BodyDocument | Determined based on Code | +| | response_exception | 响应异常 | -- | errmsg in BodyDocument | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +**Metrics field mapping table — only fields with mapping relationships are included below** + +| Name | Chinese | Request | Response | Description | +| ------------------- | --------------- | ------- | --------- | ------------------------------------------------ | +| request | 请求 | -- | -- | Number of requests | +| response | 响应 | -- | | Number of responses | +| error | 异常 | -- | -- | Client errors + Server errors | +| client_error | 客户端异常 | -- | -- | -- | +| server_error | 服务端异常 | -- | `ERR` message | See description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Errors / Responses | +| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | +| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | + +# Memcached + +By parsing the [Memcached](https://github.com/memcached/memcached/blob/master/doc/protocol.txt) protocol, the fields of Memcached Request / Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationship is shown in the table below: + +**Tag field mapping table — only fields with mapping relationships are included below** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------- | -------------------------- | --------------------------------- | ---------------------------- | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_type | 请求类型 | First word in Payload | -- | -- | +| | request_domain | 请求域名 | -- | -- | -- | +| | request_resource | 请求资源 | First line in Payload (\r\n) | -- | -- | +| | request_id | 请求 ID | -- | -- | -- | +| | endpoint | 端点 | -- | -- | -- | +| Resp. | response_code | 响应码 | -- | -- | -- | +| | response_status | 响应状态 | -- | First word in Payload | -- | +| | response_exception | 响应异常 | -- | Error message in first line of Payload when exception occurs | -- | +| | response_result | 响应结果 | -- | First line in Payload | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +**Metrics field mapping table — only fields with mapping relationships are included below** + +| Name | Chinese | Request | Response | Description | +| ------------------- | --------------- | ------- | --------- | ------------------------------------------------ | +| request | 请求 | -- | -- | Number of requests | +| response | 响应 | -- | | Number of responses | +| error | 异常 | -- | -- | Client errors + Server errors | +| client_error | 客户端异常 | -- | -- | -- | +| server_error | 服务端异常 | -- | `ERR` message | See description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Errors / Responses | +| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | +| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | \ No newline at end of file diff --git a/translate/translated/05-features/01-l7-protocols/06-mq.md b/translate/translated/05-features/01-l7-protocols/06-mq.md index 00fce6d6..d31c880d 100644 --- a/translate/translated/05-features/01-l7-protocols/06-mq.md +++ b/translate/translated/05-features/01-l7-protocols/06-mq.md @@ -7,110 +7,111 @@ permalink: /features/l7-protocols/mq # Kafka -By parsing the [Kafka](https://kafka.apache.org/protocol.html#protocol_messages) protocol, the fields of Kafka Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is as follows: +By parsing the [Kafka](https://kafka.apache.org/protocol.html#protocol_messages) protocol, the fields of Kafka Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ------------------------- | ------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | request_api_version | request_api_key | -- | -| | request_type | 请求类型 | request_api_key | request_api_version | Supported [API Key List](https://kafka.apache.org/protocol.html#protocol_api_keys) | -| | request_domain | 请求域名 | topic | topic | Only for Produce, Fetch messages, take the first corresponding field | -| | request_resource | 请求资源 | $topic-$partition:$offset | $topic-$partition:$offset | Only for Produce, Fetch messages, take the first corresponding field [1][2] | -| | request_id | 请求 ID | correlation_id | correlation_id | Reference: [Using CorrelationID to Associate Req-Resp Communication Scenarios](https://cwiki.apache.org/confluence/display/KAFKA/A+Guide+To+The+Kafka+Protocol#AGuideToTheKafkaProtocol-CommonRequestandResponseStructure) | -| | endpoint | 端点 | $topic-$partition | $topic-$partition | Only for Produce, Fetch messages, take the first corresponding field | -| Resp. | response_code | 响应码 | -- | error_code | Only for Produce, Fetch, JoinGroup, LeaveGroup, SyncGroup messages | -| | response_status | 响应状态 | -- | error_code | Normal: error_code=0; Server exception: error_code!=0 | -| | response_exception | 响应异常 | -- | error_code | [English description](http://kafka.apache.org/protocol#protocol_error_codes) of error_code | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Extracted from the corresponding Header field of the first Record | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Extracted from the corresponding Header field of the first Record | -| | x_request_id | X-Request-ID | correlation_id | correlation_id | Reference: [Using CorrelationID to Associate Req-Resp Communication Scenarios](https://cwiki.apache.org/confluence/display/KAFKA/A+Guide+To+The+Kafka+Protocol#AGuideToTheKafkaProtocol-CommonRequestandResponseStructure) | -| Misc. | attribute.group_id | -- | group_id | group_id | Only for JoinGroup, LeaveGroup, SyncGroup messages | - -- [1] All partitions in the table correspond to the partition id or partition index in the Kafka protocol. -- [2] The offset of Produce is taken from base_offset in the Response, and the offset of Fetch is taken from fetch_offset in the Request. +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | ------------------------- | ------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Req. | version | Protocol Version | request_api_version | request_api_key | -- | +| | request_type | Request Type | request_api_key | request_api_version | Supported [API Key List](https://kafka.apache.org/protocol.html#protocol_api_keys) | +| | request_domain | Request Domain | topic | topic | Only for Produce, Fetch messages, takes the first corresponding field | +| | request_resource | Request Resource | $topic-$partition:$offset | $topic-$partition:$offset | Only for Produce, Fetch messages, takes the first corresponding field [1][2] | +| | request_id | Request ID | correlation_id | correlation_id | Reference: [Using CorrelationID to Associate Req-Resp Communication Scenarios](https://cwiki.apache.org/confluence/display/KAFKA/A+Guide+To+The+Kafka+Protocol#AGuideToTheKafkaProtocol-CommonRequestandResponseStructure) | +| | endpoint | Endpoint | $topic-$partition | $topic-$partition | Only for Produce, Fetch messages, takes the first corresponding field | +| Resp. | response_code | Response Code | -- | error_code | Only for Produce, Fetch, JoinGroup, LeaveGroup, SyncGroup messages [3] | +| | response_status | Response Status | -- | error_code | Normal: error_code=0; Server exception: error_code!=0 | +| | response_exception | Response Exception | -- | error_code | [English description](http://kafka.apache.org/protocol#protocol_error_codes) of error_code [3] | +| | response_result | Response Result | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Extracted from the corresponding Header field of the first Record | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Extracted from the corresponding Header field of the first Record | +| | x_request_id | X-Request-ID | correlation_id | correlation_id | Reference: [Using CorrelationID to Associate Req-Resp Communication Scenarios](https://cwiki.apache.org/confluence/display/KAFKA/A+Guide+To+The+Kafka+Protocol#AGuideToTheKafkaProtocol-CommonRequestandResponseStructure) | +| Misc. | attribute.group_id | -- | group_id | group_id | Only for JoinGroup, LeaveGroup, SyncGroup messages | + +- [1] All partitions in the table correspond to partition id or partition index in the Kafka protocol. +- [2] Produce's offset is taken from base_offset in Response, Fetch's offset is taken from fetch_offset in Request. +- [3] Currently unsupported api key types will set response_code to -2 and response_exception to "Type not yet inspected by DeepFlow". **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------------ | ------------ | ----------------------------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| session_length | 会话长度 | -- | -- | Request length + Response length | -| request_length | 请求长度 | message_size | -- | -- | -| request_length | 响应长度 | -- | message_size | -- | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client exceptions + Server exceptions | -| client_error | 客户端异常 | -- | error_code | Refer to the description of the Tag field `response_status` | -| server_error | 服务端异常 | -- | error_code | Refer to the description of the Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Exceptions / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client exceptions / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server exceptions / Responses | +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------------ | ------------ | ------------------------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | | Number of Responses | +| session_length | Session Length | -- | -- | Request length + Response length | +| request_length | Request Length | message_size | -- | -- | +| request_length | Response Length | -- | message_size | -- | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | error_code | Reference Tag field `Response Status` description | +| server_error | Server Exception | -- | error_code | Reference Tag field `response_code` description | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | # MQTT -By parsing the [MQTT](http://docs.oasis-open.org/mqtt/mqtt/v3.1.1/os/mqtt-v3.1.1-os.html) protocol, the fields of MQTT Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is as follows: +By parsing the [MQTT](http://docs.oasis-open.org/mqtt/mqtt/v3.1.1/os/mqtt-v3.1.1-os.html) protocol, the fields of MQTT Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | -------------- | --------------- | ------------------------------------------------------------------------ | -| Req. | version | 协议版本 | -- | -- | -- | -| | request_type | 请求类型 | PacketKind | -- | -- | -| | request_domain | 请求域名 | client_id | -- | -- | -| | request_resource | 请求资源 | topic | -- | -- | -| | request_id | 请求 ID | -- | -- | -- | -| | endpoint | 端点 | topic | -- | -- | -| Resp. | response_code | 响应码 | -- | code | Only `connect_ack` message retrieves the code | -| | response_status | 响应状态 | -- | code | Normal: code=0; Client exception: code=1/2/4/5; Server exception: code=3 | -| | response_exception | 响应异常 | -- | -- | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | -- | -- | -- | -| | span_id | SpanID | -- | -- | -- | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | -------------- | --------------- | ------------------------------------------------------------------------ | +| Req. | version | Protocol Version | -- | -- | -- | +| | request_type | Request Type | PacketKind | -- | -- | +| | request_domain | Request Domain | client_id | -- | -- | +| | request_resource | Request Resource | topic | -- | -- | +| | request_id | Request ID | -- | -- | -- | +| | endpoint | Endpoint | topic | -- | -- | +| Resp. | response_code | Response Code | -- | code | Only `connect_ack` messages get code | +| | response_status | Response Status | -- | code | Normal: code=0; Client exception: code=1/2/4/5; Server exception: code=3 | +| | response_exception | Response Exception | -- | -- | -- | +| | response_result | Response Result | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------------------------- | ----------------------------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client exceptions + Server exceptions | -| client_error | 客户端异常 | -- | `connect_ack` message code | Refer to the description of the Tag field `response_status` | -| server_error | 服务端异常 | -- | `connect_ack` message code | Refer to the description of the Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Exceptions / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client exceptions / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server exceptions / Responses | +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------- | ----------------------------------- | ------------------------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | | Number of Responses | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | `connect_ack` message returned code | Reference Tag field `Response Status` description | +| server_error | Server Exception | -- | `connect_ack` message returned code | Reference Tag field `response_code` description | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | # AMQP -By parsing the [AMQP](https://www.rabbitmq.com/specification.html) protocol (i.e., the main protocol of [RabbitMQ](https://www.rabbitmq.com/protocols.html)), the fields of AMQP Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is as follows: +By parsing the [AMQP](https://www.rabbitmq.com/specification.html) protocol (i.e., the main protocol of [RabbitMQ](https://www.rabbitmq.com/protocols.html)), the fields of AMQP Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ----------------------------- | ---------------- | ------------------------------ | -| Req. | version | 协议版本 | version | -- | 0-9-1 | -| | request_type | 请求类型 | class_id.method_id | -- | e.g., Channel.OpenOK | -| | request_domain | 请求域名 | vhost | -- | -- | -| | request_resource | 请求资源 | exchange.routing_key or queue | -- | -- | -| | request_id | 请求 ID | -- | -- | -- | -| | endpoint | 端点 | exchange.routing_key or queue | -- | -- | -| Resp. | response_code | 响应码 | -- | method_id | OpenOK | -| | response_status | 响应状态 | -- | -- | All considered normal | -| | response_exception | 响应异常 | -- | -- | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Custom field in Content Header | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Custom field in Content Header | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | - -Note: Due to protocol characteristics, currently only AMQP protocols established after the agent starts are supported. - -Additionally, the following are one-way messages and will be directly saved as type=session call logs: +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | ----------------------------- | ---------------- | ------------------------------- | +| Req. | version | Protocol Version | version | -- | 0-9-1 | +| | request_type | Request Type | class_id.method_id | -- | Example: Channel.OpenOK | +| | request_domain | Request Domain | vhost | -- | -- | +| | request_resource | Request Resource | exchange.routing_key or queue | -- | -- | +| | request_id | Request ID | -- | -- | -- | +| | endpoint | Endpoint | exchange.routing_key or queue | -- | -- | +| Resp. | response_code | Response Code | -- | method_id | OpenOK | +| | response_status | Response Status | -- | -- | All considered normal | +| | response_exception | Response Exception | -- | -- | -- | +| | response_result | Response Result | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Custom fields in Content Header | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Custom fields in Content Header | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +Note: Limited by protocol characteristics, currently only supports identifying AMQP protocols that establish connections after the agent starts. + +Additionally, the following are unidirectional messages that will be directly saved as type=session call logs: - Connection.Blocked (`s->c`) - Connection.Unblocked (`s->c`) @@ -122,127 +123,127 @@ Additionally, the following are one-way messages and will be directly saved as t - Content-Header (`both`) - Content-Body (`both`) - Protocol-Header (`c->s`) - The following messages may or may not have ACK, DeepFlow uniformly ignores their responses (because the ACK does not contain key information, and since ACK is not stable, there is no need to calculate latency): + While the following messages may or may not have ACK, DeepFlow uniformly ignores their responses (because ACK contains no critical information, and since ACK is not stable, there's no need to calculate latency): - Basic.Publish (`c->s`) - Basic.Deliver (`s->c`) **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------- | ------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | -- | Number of Responses | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client exceptions + Server exceptions | -| client_error | 客户端异常 | -- | -- | -- | -| server_error | 服务端异常 | -- | -- | -- | -| error_ratio | 异常比例 | -- | -- | Exceptions / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client exceptions / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server exceptions / Responses | +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------- | -------- | ----------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | -- | Number of Responses | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | -- | -- | +| server_error | Server Exception | -- | -- | -- | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | # OpenWire -By parsing the [OpenWire](https://activemq.apache.org/openwire) protocol (i.e., the default protocol of [ActiveMQ](https://activemq.apache.org/protocols)), the fields of OpenWire Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is as follows: +By parsing the [OpenWire](https://activemq.apache.org/openwire) protocol (i.e., the default protocol of [ActiveMQ](https://activemq.apache.org/protocols)), the fields of OpenWire Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ---------------- | ------------------ | -------------------------------------------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | version | -- | -- | -| | request_type | 请求类型 | OpenWireCommand | -- | -- | -| | request_domain | 请求域名 | broker_url | -- | -- | -| | request_resource | 请求资源 | topic | -- | -- | -| | request_id | 请求 ID | command_id | correlation_id [1] | The correspondence between request and response is detailed in [2] | -| | endpoint | 端点 | topic | -- | -- | -| Resp. | response_code | 响应码 | -- | -- | -- | -| | response_status | 响应状态 | -- | -- | Normal: no error message; Server exception: has error message | -| | response_exception | 响应异常 | -- | error message | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | -- | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | -- | -| | x_request_id | X-Request-ID | correlation_id | correlation_id | Reference: [CorrelationID in ActiveMQ](https://activemq.apache.org/how-should-i-implement-request-response-with-jms) | -| Misc. | -- | -- | -- | -- | -- | - -- [1] Note the distinction from the correlation_id field corresponding to x_request_id below, they are two different fields -- [2] When the response_required of the request is true, the correlation_id field of the corresponding response should be consistent with the command_id of the request +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | ---------------- | ------------------ | -------------------------------------------------------------------------------------------------------------------- | +| Req. | version | Protocol Version | version | -- | -- | +| | request_type | Request Type | OpenWireCommand | -- | -- | +| | request_domain | Request Domain | broker_url | -- | -- | +| | request_resource | Request Resource | topic | -- | -- | +| | request_id | Request ID | command_id | correlation_id [1] | Request and response correspondence details see [2] | +| | endpoint | Endpoint | topic | -- | -- | +| Resp. | response_code | Response Code | -- | -- | -- | +| | response_status | Response Status | -- | -- | Normal: no error message; Server exception: has error message | +| | response_exception | Response Exception | -- | error message | -- | +| | response_result | Response Result | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | -- | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | -- | +| | x_request_id | X-Request-ID | correlation_id | correlation_id | Reference: [CorrelationID in ActiveMQ](https://activemq.apache.org/how-should-i-implement-request-response-with-jms) | +| Misc. | -- | -- | -- | -- | -- | + +- [1] Note the distinction from the correlation_id field corresponding to x_request_id below, these are two different fields +- [2] When request's response_required is true, the correlation_id field of the corresponding response should match the command_id of the request **Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------- | ------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client exceptions + Server exceptions | -| client_error | 客户端异常 | -- | -- | -- | -| server_error | 服务端异常 | -- | -- | -- | -| error_ratio | 异常比例 | -- | -- | Exceptions / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client exceptions / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server exceptions / Responses | +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------- | -------- | ----------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | | Number of Responses | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | -- | -- | +| server_error | Server Exception | -- | -- | -- | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | # NATS -By parsing the [NATS](https://docs.nats.io/reference/reference-protocols/nats-protocol) protocol, the fields of NATS Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is as follows: +By parsing the [NATS](https://docs.nats.io/reference/reference-protocols/nats-protocol) protocol, the fields of NATS Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: **Tag Field Mapping Table, the following table only includes fields with mapping relationships** -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ---------------- | ---------------- | --------------------------------------------- | -| Req. | version | 协议版本 | version | -- | Using version in INFO | -| | request_type | 请求类型 | NatsMessage | -- | Such as INFO, SUB, PUB, MSG | -| | request_domain | 请求域名 | server_name | -- | Using server_name in INFO | -| | request_resource | 请求资源 | subject | -- | -- | -| | request_id | 请求 ID | -- | -- | -- | -| | endpoint | 端点 | subject | -- | Only the part before the first `.` in subject | -| Resp. | response_code | 响应码 | -- | -- | -- | -| | response_status | 响应状态 | -- | -- | All considered normal | -| | response_exception | 响应异常 | -- | -- | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | -| | x_request_id | X-Request-ID | -- | -- | -- | -| Misc. | -- | -- | -- | -- | -- | - -Note, except for the two pairs of messages Info/Connect and Ping/Pong, other messages are one-way messages and will be directly saved as type=session call logs: - -**Metrics Field Mapping Table: The following table includes only the fields that have a mapping relationship** - -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------- | ----------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client Errors + Server Errors | -| client_error | 客户端异常 | -- | -- | -- | -| server_error | 服务端异常 | -- | -- | -- | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Errors / Responses | +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | ---------------- | ---------------- | --------------------------------------------- | +| Req. | version | Protocol Version | version | -- | Uses version from INFO | +| | request_type | Request Type | NatsMessage | -- | Such as INFO, SUB, PUB, MSG | +| | request_domain | Request Domain | server_name | -- | Uses server_name from INFO | +| | request_resource | Request Resource | subject | -- | -- | +| | request_id | Request ID | -- | -- | -- | +| | endpoint | Endpoint | subject | -- | Only the part before the first `.` in subject | +| Resp. | response_code | Response Code | -- | -- | -- | +| | response_status | Response Status | -- | -- | All considered normal | +| | response_exception | Response Exception | -- | -- | -- | +| | response_result | Response Result | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +Note that except for Info/Connect, Ping/Pong these two groups of messages, other messages are all unidirectional and will be directly saved as type=session call logs: + +**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** + +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------- | -------- | ----------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | | Number of Responses | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | -- | -- | +| server_error | Server Exception | -- | -- | -- | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | # Pulsar -By parsing the [Pulsar](https://pulsar.apache.org/docs/3.2.x/client-libraries-python/) protocol, the fields of Pulsar Request / Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationships are shown in the following table: - -**Tag Field Mapping Table: The following table includes only the fields that have a mapping relationship** - -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | ------------ | ------------------- | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Req. | version | 协议版本 | protocol_version | -- | Taken from the smaller of CommandConnect and CommandConnected | -| | request_type | 请求类型 | command | -- | -- | -| | request_domain | 请求域名 | proxy_to_broker_url | -- | Taken from CommandConnect | -| | request_resource | 请求资源 | topic | -- | The content after the last `/` in the protocol topic | -| | request_id | 请求 ID | request_id | -- | For Send/SendError/SendReceipt, since these commands do not have request_id, concatenate the lower 16 bits of producer_id and sequence_id as the request ID | -| | endpoint | 端点 | topic | -- | -- | -| Resp. | response_code | 响应码 | -- | code | -- | -| | response_status | 响应状态 | -- | status | -- | -| | response_exception | 响应异常 | -- | exception | -- | -| | response_result | 响应结果 | -- | -- | -- | -| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | -| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | -| | x_request_id | X-Request-ID | x_request_id | x_request_id | -- | -| Misc. | -- | -- | -- | -- | -- | - -Note that the following are unidirectional messages, which will be directly saved as `type=session` call logs: +By parsing the [Pulsar](https://pulsar.apache.org/docs/3.2.x/client-libraries-python/) protocol, the fields of Pulsar Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: + +**Tag Field Mapping Table, the following table only includes fields with mapping relationships** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | ------------------- | ---------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ | +| Req. | version | Protocol Version | protocol_version | -- | Takes the smaller of CommandConnect and CommandConnected | +| | request_type | Request Type | command | -- | -- | +| | request_domain | Request Domain | proxy_to_broker_url | -- | In CommandConnect | +| | request_resource | Request Resource | topic | -- | Takes the content after the last / in the protocol topic | +| | request_id | Request ID | request_id | -- | For Send/SendError/SendReceipt, since the command has no request_id, takes producer_id and the lower 16 bits of sequence_id concatenated as request ID | +| | endpoint | Endpoint | topic | -- | -- | +| Resp. | response_code | Response Code | -- | code | -- | +| | response_status | Response Status | -- | status | -- | +| | response_exception | Response Exception | -- | exception | -- | +| | response_result | Response Result | -- | -- | -- | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Extracted from NATS headers in HMSG, HPUB | +| | x_request_id | X-Request-ID | x_request_id | x_request_id | -- | +| Misc. | -- | -- | -- | -- | -- | + +Note that the following are unidirectional messages that will be directly saved as type=session call logs: - Ack - Flow @@ -257,50 +258,91 @@ Note that the following are unidirectional messages, which will be directly save - WatchTopicListClose - TopicMigrated -**Metrics Field Mapping Table: The following table includes only the fields that have a mapping relationship** +**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------- | ----------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client Errors + Server Errors | -| client_error | 客户端异常 | -- | -- | -- | -| server_error | 服务端异常 | -- | -- | -- | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Errors / Responses | +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------- | -------- | ----------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | | Number of Responses | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | -- | -- | +| server_error | Server Exception | -- | -- | -- | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | # ZMTP -By parsing the [ZMTP](https://rfc.zeromq.org/spec/23/) protocol (i.e., the messaging transport protocol used by ZeroMQ), the fields of ZMTP Request / Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationships are shown in the following table: - -**Tag Field Mapping Table: The following table includes only the fields that have a mapping relationship** - -| Category | Name | Chinese | Request Header | Response Header | Description | -| -------- | ------------------ | -------- | -------------- | --------------- | ------------------------------------------------------ | -| Req. | version | 协议版本 | version | -- | -- | -| | request_type | 请求类型 | frame_type | -- | -- | -| | request_domain | 请求域名 | subscription | -- | Only when socket type is PUB/SUB/XPUB/XSUB | -| | request_resource | 请求资源 | subscription | -- | Only when socket type is PUB/SUB/XPUB/XSUB | -| Resp. | response_code | 响应码 | -- | -- | -- | -| | response_status | 响应状态 | -- | -- | Normal: No error message; Exception: Has error message | -| | response_exception | 响应异常 | -- | error message | -- | -| Misc. | -- | -- | -- | -- | -- | - -- In the ZMTP protocol, when one end socket is REQ/REP, a request message must wait for the previous request to get a response before sending, and requests and responses will be aggregated into one session -- Other types are currently only recognized as unidirectional messages and will be directly saved as `type=session` call logs - -**Metrics Field Mapping Table: The following table includes only the fields that have a mapping relationship** - -| Name | Chinese | Request | Response | Description | -| ------------------ | -------------- | ------- | -------- | ----------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | | Number of Responses | -| log_count | 日志总量 | -- | -- | -- | -| error | 异常 | -- | -- | Client Errors + Server Errors | -| client_error | 客户端异常 | -- | -- | -- | -| server_error | 服务端异常 | -- | -- | -- | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Errors / Responses | +By parsing the [ZMTP](https://rfc.zeromq.org/spec/23/) protocol (i.e., the message transport protocol used by ZeroMQ), the fields of ZMTP Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: + +**Tag Field Mapping Table, the following table only includes fields with mapping relationships** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | -------------- | --------------- | ------------------------------------------------------ | +| Req. | version | Protocol Version | version | -- | -- | +| | request_type | Request Type | frame_type | -- | -- | +| | request_domain | Request Domain | subscription | -- | Only when socket type is PUB/SUB/XPUB/XSUB | +| | request_resource | Request Resource | subscription | -- | Only when socket type is PUB/SUB/XPUB/XSUB | +| Resp. | response_code | Response Code | -- | -- | -- | +| | response_status | Response Status | -- | -- | Normal: no error message; Exception: has error message | +| | response_exception | Response Exception | -- | error message | -- | +| Misc. | -- | -- | -- | -- | -- | + +- In the ZMTP protocol, only when one end socket is REQ/REP, request messages must wait for the previous request to receive a response before initiating, and the request and response will be aggregated into a session +- Other types are currently only identified as unidirectional messages and will be directly saved as type=session call logs + +**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** + +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------- | -------- | ----------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | | Number of Responses | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | -- | -- | +| server_error | Server Exception | -- | -- | -- | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | + +# RocketMQ + +By parsing the [RocketMQ](https://rocketmq.apache.org/docs/4.x/) protocol, the fields of RocketMQ Request / Response are mapped to the corresponding fields in l7_flow_log. The mapping relationships are shown in the following table: + +**Tag Field Mapping Table, the following table only includes fields with mapping relationships** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------------ | ---------------------------------------- | ---------------- | --------------------------------------------------------------------------------------------------------------------- | +| Req. | version | Protocol Version | version | -- | -- | +| | request_type | Request Type | code | -- | -- | +| | request_domain | Request Domain | extFields:producerGroup \| consumerGroup | -- | Mainly for SEND and PULL, other messages not yet fully supplemented | +| | request_resource | Request Resource | extFields:topic | -- | Mainly for SEND and PULL, other messages not yet fully supplemented | +| | request_id | Request ID | opaque | -- | -- | +| | endpoint | Endpoint | extFields:topic & queueId | -- | -- | +| Resp. | response_code | Response Code | -- | code | -- | +| | response_status | Response Status | -- | code | 0 represents normal, non-0 represents various exceptions | +| | response_exception | Response Exception | -- | remark | -- | +| | response_result | Response Result | -- | body | JSON serialization format corresponds to all JSON string data, ROCKETMQ format corresponds to the body string therein | +| Trace | trace_id | TraceID | traceparent, sw8 | traceparent, sw8 | Extracted from extFields or properties field in bodyData | +| | span_id | SpanID | traceparent, sw8 | traceparent, sw8 | Extracted from extFields or properties field in bodyData | +| | x_request_id | X-Request-ID | UNIQ_KEY | KEY | Extracted from extFields or properties field in bodyData | +| Misc. | -- | -- | -- | -- | -- | + +- In the RocketMQ protocol, bit0 of the flag field identifies whether it's a request or response, while bit1 identifies whether it's a unidirectional request (no response needed) +- For non-unidirectional requests like SEND_MESSAGE, PULL_MESSAGE and their corresponding responses, they are aggregated into a Session one-to-one according to opaque +- For unidirectional requests like UPDATE_CONSUMER_OFFSET, they are set as Session type individually, no aggregation needed + +**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** + +| Name | Chinese | Request | Response | Description | +| ------------------ | ---------------------- | ------- | -------- | ----------------------------------- | +| request | Request | -- | -- | Number of Requests | +| response | Response | -- | -- | Number of Responses | +| log_count | Total Logs | -- | -- | -- | +| error | Exception | -- | -- | Client exception + Server exception | +| client_error | Client Exception | -- | -- | -- | +| server_error | Server Exception | -- | -- | -- | +| error_ratio | Exception Ratio | -- | -- | Exception / Response | +| client_error_ratio | Client Exception Ratio | -- | -- | Client exception / Response | +| server_error_ratio | Server Exception Ratio | -- | -- | Server exception / Response | diff --git a/translate/translated/05-features/01-l7-protocols/07-network.md b/translate/translated/05-features/01-l7-protocols/07-network.md index 0feaab8e..943e7725 100644 --- a/translate/translated/05-features/01-l7-protocols/07-network.md +++ b/translate/translated/05-features/01-l7-protocols/07-network.md @@ -7,37 +7,72 @@ permalink: /features/l7-protocols/network # DNS -By parsing the [DNS](https://www.ietf.org/rfc/rfc1035.txt) protocol, the fields of DNS Request/Response are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: +By parsing the [DNS](https://www.ietf.org/rfc/rfc1035.txt) protocol, the fields of DNS Request / Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationship is shown in the table below: -**Tag Field Mapping Table, the following table only includes fields with mapping relationships** +**Tag field mapping table — only fields with mapping relationships are included** | Category | Name | Chinese | Request Header | Response Header | Description | | -------- | ------------------ | ------------- | -------------- | --------------- | --------------------------------------------------------------------------------------------- | | Req. | version | 协议版本 | -- | -- | -- | | | request_type | 请求类型 | QTYPE | -- | -- | -| | request_domain | 请求域名 | QNAME | -- | Only has value when querying IPv4 or IPv6 addresses | +| | request_domain | 请求域名 | QNAME | -- | Only populated when querying IPv4 or IPv6 addresses | | | request_resource | 请求资源 | QNAME | -- | -- | | | request_id | 请求 ID | ID | ID | -- | | | endpoint | 端点 | QNAME | -- | -- | | Resp. | response_code | 响应码 | -- | RCODE | -- | -| | response_status | 响应状态 | -- | RCODE | Normal: RCODE=0x0; Client Error: RCODE=0x1/0x3; Server Error: others | -| | response_exception | 响应异常 | -- | RCODE | Description of RCODE, refer to [RFC 2929 Section 2.3](https://www.rfc-editor.org/rfc/rfc2929#section-2.3) | +| | response_status | 响应状态 | -- | RCODE | Normal: RCODE=0x0; Client error: RCODE=0x1/0x3; Server error: others | +| | response_exception | 响应异常 | -- | RCODE | Description of RCODE, see [RFC 2929 Section 2.3](https://www.rfc-editor.org/rfc/rfc2929#section-2.3) | | | response_result | 响应结果 | -- | RDATA | -- | | Trace | trace_id | TraceID | -- | -- | -- | | | span_id | SpanID | -- | -- | -- | | | x_request_id | X-Request-ID | -- | -- | -- | | Misc. | -- | -- | -- | -- | -- | -**Metrics Field Mapping Table, the following table only includes fields with mapping relationships** - -| Name | Chinese | Request | Response | Description | -| ------------------- | ----------------- | ------- | -------- | -------------------------------------- | -| request | 请求 | -- | -- | Number of Requests | -| response | 响应 | -- | -- | Number of Responses | -| log_count | 日志总量 | -- | -- | Number of Request Log lines | -| error | 异常 | -- | -- | Client Errors + Server Errors | -| client_error | 客户端异常 | -- | RCODE | Refer to the description of Tag field `response_code` | -| server_error | 服务端异常 | -- | RCODE | Refer to the description of Tag field `response_code` | -| error_ratio | 异常比例 | -- | -- | Errors / Responses | -| client_error_ratio | 客户端异常比例 | -- | -- | Client Errors / Responses | -| server_error_ratio | 服务端异常比例 | -- | -- | Server Errors / Responses | +**Metrics field mapping table — only fields with mapping relationships are included** + +| Name | Chinese | Request | Response | Description | +| ------------------ | --------------- | ------- | -------- | ------------------------------------------------------ | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | -- | -- | Number of Responses | +| log_count | 日志总量 | -- | -- | Number of Request Log entries | +| error | 异常 | -- | -- | Client errors + Server errors | +| client_error | 客户端异常 | -- | RCODE | See the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | RCODE | See the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Errors / Responses | +| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | +| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | + +# Ping + +By parsing Echo messages in the [ICMP](https://www.rfc-editor.org/rfc/rfc792) protocol, the fields of Echo Request / Response are mapped to the corresponding fields in `l7_flow_log`. The mapping relationship is shown in the table below: + +**Tag field mapping table — only fields with mapping relationships are included** + +| Category | Name | Chinese | Request Header | Response Header | Description | +| -------- | ------------------ | ------------- | --------------- | --------------- | ------------------------------------------------------------------ | +| Req. | version | 协议版本 | -- | -- | -- | +| | request_resource | 请求资源 | Identifier | -- | -- | +| | request_id | 请求 ID | Sequence Number | -- | -- | +| | endpoint | 端点 | -- | -- | -- | +| Resp. | response_code | 响应码 | -- | -- | -- | +| | response_status | 响应状态 | -- | -- | If a response is received, record as normal; if not, record as timeout | +| | response_exception | 响应异常 | -- | -- | -- | +| | response_result | 响应结果 | -- | -- | -- | +| Trace | trace_id | TraceID | -- | -- | -- | +| | span_id | SpanID | -- | -- | -- | +| | x_request_id | X-Request-ID | -- | -- | -- | +| Misc. | -- | -- | -- | -- | -- | + +**Metrics field mapping table — only fields with mapping relationships are included** + +| Name | Chinese | Request | Response | Description | +| ------------------ | --------------- | ------- | -------- | ------------------------------------------------------ | +| request | 请求 | -- | -- | Number of Requests | +| response | 响应 | -- | -- | Number of Responses | +| log_count | 日志总量 | -- | -- | Number of Request Log entries | +| error | 异常 | -- | -- | Client errors + Server errors | +| client_error | 客户端异常 | -- | -- | See the description of Tag field `response_code` | +| server_error | 服务端异常 | -- | -- | See the description of Tag field `response_code` | +| error_ratio | 异常比例 | -- | -- | Errors / Responses | +| client_error_ratio | 客户端异常比例 | -- | -- | Client errors / Responses | +| server_error_ratio | 服务端异常比例 | -- | -- | Server errors / Responses | \ No newline at end of file diff --git a/translate/translated/05-features/01-l7-protocols/08-otel.md b/translate/translated/05-features/01-l7-protocols/08-otel.md index 947e4d54..fac7193f 100644 --- a/translate/translated/05-features/01-l7-protocols/08-otel.md +++ b/translate/translated/05-features/01-l7-protocols/08-otel.md @@ -5,60 +5,60 @@ permalink: /features/l7-protocols/otel > This document was translated by ChatGPT -By parsing the OpenTelemetry protocol, the fields of the OpenTelemetry protocol data structure are mapped to the corresponding fields in l7_flow_log. The mapping relationship is shown in the table below: +By parsing the OpenTelemetry protocol, the fields in the OpenTelemetry protocol data structure are mapped to the corresponding fields in `l7_flow_log`. The mapping relationship is shown in the table below: -**Tag Field Mapping Table, the following table only includes fields that have a mapping relationship** +**Tag Field Mapping Table — only fields with mapping relationships are included below** -| Name | Chinese | OpenTelemetry Data Structure | Description | -| ------------------- | ------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------ | -| start_time | 开始时间 | span.start_time_unix_nano | -- | -| end_time | 结束时间 | span.end_time_unix_nano | -- | -| protocol | 网络协议 | span.attribute.net.transport | Mapped to the corresponding enum value | -| attributes | Misc.butes | resource./span.attributes | -- | -| ip | IP 地址 | span.attribute.app.host.ip/attribute.net.peer.ip | Detailed explanation in the following paragraphs | -| l7_protocol | 应用协议 | span.attribute.http.scheme/db.system/rpc.system/messaging.system/messaging.protocol | Mapped to the corresponding enum value, any attribute starting with http is considered as HTTP protocol | -| l7_protocol_str | 应用协议 | span.attribute.http.scheme/db.system/rpc.system/messaging.system/messaging.protocol | If span.attribute.http.scheme exists, read it; if not, but l7_protocol is HTTP, default to HTTP | -| version | 协议版本 | span.attribute.http.flavor | -- | -| type | 日志类型 | 会话 | -- | -| request_type | 请求类型 | span.attribute.http.method/db.operation/rpc.method | -- | -| request_domain | 请求域名 | span.attribute.http.host/db.connection_string | -- | -| request_resource | 请求资源 | attribute.http.target/db.statement/messaging.url/rpc.service | If span.attribute.http.target exists, read it; if not, extract from http.url, only extracting the call information after the domain name | +| Name | Chinese | OpenTelemetry Data Structure | Description | +| ------------------- | ------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------- | +| start_time | 开始时间 | span.start_time_unix_nano | -- | +| end_time | 结束时间 | span.end_time_unix_nano | -- | +| protocol | 网络协议 | span.attribute.net.transport | Mapped to the corresponding enum value | +| attributes | 属性标签 | resource./span.attributes | -- | +| ip | IP 地址 | span.attribute.app.host.ip/attribute.net.peer.ip | See detailed explanation in the following section | +| l7_protocol | 应用协议 | span.attribute.http.scheme/db.system/rpc.system/messaging.system/messaging.protocol | Mapped to the corresponding enum value; if any attribute starts with `http`, it is considered HTTP protocol | +| l7_protocol_str | 应用协议 | span.attribute.http.scheme/db.system/rpc.system/messaging.system/messaging.protocol | If `span.attribute.http.scheme` exists, read it; if not but `l7_protocol` is HTTP, default to HTTP | +| version | 协议版本 | span.attribute.http.flavor | -- | +| type | 日志类型 | 会话 | -- | +| request_type | 请求类型 | span.attribute.http.method/db.operation/rpc.method | -- | +| request_domain | 请求域名 | span.attribute.http.host/db.connection_string | -- | +| request_resource | 请求资源 | attribute.http.target/db.statement/messaging.url/rpc.service | If `span.attribute.http.target` exists, read it; if not, extract from `http.url` after the domain | | request_id | 请求 ID | -| response_status | 响应状态 | Response code = span.attribute.http.status_code, refer to HTTP protocol definition; Response code = span.status.code, unknown: STATUS_CODE_UNSET; normal: STATUS_CODE_OK; server error: STATUS_CODE_ERROR | -- | -| response_code | 响应码 | span.attribute.http.status_code/span.status.code | Prefer span.attribute.http.status_code | -| response_exception | 响应异常 | Response code = span.attribute.http.status_code, refer to HTTP protocol definition; Response code = span.status.code, corresponding to `span.status.message` | -- | -| service_name | 服务名称 | resource./span.attribute.service.name | -- | -| service_instance_id | 服务实例 | resource./span.attribute.service.instance.id | -- | -| endpoint | 端点 | span.name | -- | -| trace_id | TraceID | span.trace_id/attribute.sw8.trace_id | Prefer attribute.sw8.trace_id | -| span_id | SpanID | span.span_id/attribute.sw8.segment_id-attribute.sw8.span_id | Prefer attribute.sw8.segment_id-attribute.sw8.span_id | -| parent_span_id | ParentSpanID | span.parent_span_id/attribute.sw8.segment_id-attribute.sw8.parent_span_id | Prefer attribute.sw8.segment_id-attribute.sw8.parent_span_id | -| span_kind | Span 类型 | span.span_kind | -- | -| events | 事件 | span.events | Saved as a JSON formatted string | -| observation_point | 观测点 | span.spankind.SPAN_KIND_CLIENT/SPAN_KIND_PRODUCER: client application (c-app); span.spankind.SPAN_KIND_SERVER/SPAN_KIND_CONSUMER: server application (s-app); span.spankind.SPAN_KIND_UNSPECIFIED/SPAN_KIND_INTERNAL: application (app) | -- | +| response_status | 响应状态 | Response code = span.attribute.http.status_code (refer to HTTP protocol definition); Response code = span.status.code, Unknown: STATUS_CODE_UNSET; OK: STATUS_CODE_OK; Server error: STATUS_CODE_ERROR | -- | +| response_code | 响应码 | span.attribute.http.status_code/span.status.code | Prefer `span.attribute.http.status_code` | +| response_exception | 响应异常 | Response code = span.attribute.http.status_code (refer to HTTP protocol definition); Response code = span.status.code, then use `span.status.message` | -- | +| service_name | 服务名称 | resource./span.attribute.service.name | -- | +| service_instance_id | 服务实例 | resource./span.attribute.service.instance.id | -- | +| endpoint | 端点 | span.name | -- | +| trace_id | TraceID | resource.attribute.sw8.trace_id/span.attribute.sw8.trace_id/span.trace_id | Priority: resource.attribute.sw8.trace_id > span.attribute.sw8.trace_id > span.trace_id | +| span_id | SpanID | span.span_id/attribute.sw8.segment_id-attribute.sw8.span_id | Prefer `attribute.sw8.segment_id-attribute.sw8.span_id` | +| parent_span_id | ParentSpanID | span.parent_span_id/attribute.sw8.segment_id-attribute.sw8.parent_span_id | Prefer `attribute.sw8.segment_id-attribute.sw8.parent_span_id` | +| span_kind | Span 类型 | span.span_kind | -- | +| events | 事件 | span.events | Saved as a JSON-formatted string | +| observation_point | 观测点 | span.spankind.SPAN_KIND_CLIENT/SPAN_KIND_PRODUCER: client application (c-app); span.spankind.SPAN_KIND_SERVER/SPAN_KIND_CONSUMER: server application (s-app); span.spankind.SPAN_KIND_UNSPECIFIED/SPAN_KIND_INTERNAL: application (app) | -- | - observation_point = c-app - - span.attribute.app.host.ip corresponds to ip_0; all others correspond to ip_1 - - Obtain the IP address of the current application (otel-agent) corresponding to the previous level (i.e., the source of the Span) through a [k8s attributes processor plugin](https://pkg.go.dev/github.com/open-telemetry/opentelemetry-collector-contrib/processor/k8sattributesprocessor#section-readme), for example: if the Span is generated by a POD, get the IP of the POD; if the Span is generated by a process deployed on a virtual machine, get the IP of the virtual machine - - span.attribute.net.peer.ip corresponds to ip_1; all others correspond to ip_0 + - `span.attribute.app.host.ip` corresponds to `ip_0`; all others correspond to `ip_1` + - Use a [k8s attributes processor plugin](https://pkg.go.dev/github.com/open-telemetry/opentelemetry-collector-contrib/processor/k8sattributesprocessor#section-readme) to obtain the IP address of the upper-level source of the current application (otel-agent), i.e., the source of the Span. For example: if the Span is generated by a POD, get the POD's IP; if the Span is generated by a process deployed on a virtual machine, get the VM's IP. + - `span.attribute.net.peer.ip` corresponds to `ip_1`; all others correspond to `ip_0` -**Metrics Field Mapping Table, the following table only includes fields that have a mapping relationship** +**Metrics Field Mapping Table — only fields with mapping relationships are included below** -| Name | Chinese | OpenTelemetry Data Structure | Description | -| ----------------------------------------------- | --------------- | -------------------------------------------------------------------- | ---------------------------------- | -| request | 请求 | Number of Spans | -- | -| response | 响应 | Number of Spans | -- | -| session_length | 会话长度 | | Request length + Response length | -| request_length | 请求长度 | span.attribute.http.request_content_length | -- | -| request_length | 响应长度 | span.attribute.http.response_content_length | -- | -| sql_affected_rows | SQL 影响行数 | span.attribute.db.cassandra.page_size | -- | -| log_count | 日志总量 | Number of Spans | Number of Request Log lines | -| error | 异常 | -- | Client error + Server error | -| client_error | 客户端异常 | span.attribute.http.status_code/span.status.code | Refer to the description of `response_code` in Tag Field | -| server_error | 服务端异常 | span.attribute.http.status_code/span.status.code | Refer to the description of `response_code` in Tag Field | -| error_ratio | 异常比例 | -- | Error / Response | -| client_error_ratio | 客户端异常比例 | -- | Client error / Response | -| server_error_ratio | 服务端异常比例 | -- | Server error / Response | -| message.uncompressed_size | -- | span.attribute.message.uncompressed_size | -- | -| messaging.message_payload_size_bytes | -- | span.attribute.messaging.message_payload_size_bytes | -- | -| messaging.message_payload_compressed_size_bytes | -- | span.attribute.messaging.message_payload_compressed_size_bytes | -- | \ No newline at end of file +| Name | Chinese | OpenTelemetry Data Structure | Description | +| ----------------------------------------------- | --------------- | --------------------------------------------------------------------- | ------------------------------------ | +| request | 请求 | Number of Spans | -- | +| response | 响应 | Number of Spans | -- | +| session_length | 会话长度 | | Request length + Response length | +| request_length | 请求长度 | span.attribute.http.request_content_length | -- | +| request_length | 响应长度 | span.attribute.http.response_content_length | -- | +| sql_affected_rows | SQL 影响行数 | span.attribute.db.cassandra.page_size | -- | +| log_count | 日志总量 | Number of Spans | Number of request log lines | +| error | 异常 | -- | Client errors + Server errors | +| client_error | 客户端异常 | span.attribute.http.status_code/span.status.code | See `response_code` in Tag fields | +| server_error | 服务端异常 | span.attribute.http.status_code/span.status.code | See `response_code` in Tag fields | +| error_ratio | 异常比例 | -- | Errors / Responses | +| client_error_ratio | 客户端异常比例 | -- | Client errors / Responses | +| server_error_ratio | 服务端异常比例 | -- | Server errors / Responses | +| message.uncompressed_size | -- | span.attribute.message.uncompressed_size | -- | +| messaging.message_payload_size_bytes | -- | span.attribute.messaging.message_payload_size_bytes | -- | +| messaging.message_payload_compressed_size_bytes | -- | span.attribute.messaging.message_payload_compressed_size_bytes | -- | \ No newline at end of file diff --git a/translate/translated/05-features/01-l7-protocols/09-skywalking.md b/translate/translated/05-features/01-l7-protocols/09-skywalking.md new file mode 100644 index 00000000..c239465d --- /dev/null +++ b/translate/translated/05-features/01-l7-protocols/09-skywalking.md @@ -0,0 +1,38 @@ +--- +title: SkyWalking +permalink: /features/l7-protocols/skywalking +--- + +> This document was translated by ChatGPT + +By parsing the SkyWalking sw8 protocol, the fields in the SkyWalking protocol data structure are mapped to the corresponding fields in `l7_flow_log`. The mapping relationship is shown in the table below: + +**Tag field mapping table — the table below only includes fields that have a mapping relationship** + +| Name | Chinese | SkyWalking Data Structure | Description | +| ------------------ | ------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------- | +| start_time | 开始时间 | span.startTime | -- | +| end_time | 结束时间 | span.endTime | -- | +| protocol | 网络协议 | TCP | Fixed enumeration value | +| attributes | 标签 | span.tags | -- | +| ip | IP 地址 | -- | Obtained from the upstream SkyWalking Agent data source | +| l7_protocol | 应用协议 | span.tags.[db.type/http.scheme/db.system/rpc.system/messaging.system/messaging.protocol] | If there are any tags starting with `http.`, mark as HTTP protocol; otherwise, try to obtain from span.tags | +| l7_protocol_str | 应用协议 | span.tags.[db.type/http.scheme/db.system/rpc.system/messaging.system/messaging.protocol] | First try to convert from l7_protocol to a description; if conversion fails, record directly as tag value | +| version | 协议版本 | span.tags.http.flavor | -- | +| type | 日志类型 | SESSION | Fixed enumeration value | +| request_type | 请求类型 | span.tags.[http.method/cache.cmd/db.operation/rpc.method] | -- | +| request_domain | 请求域名 | span.tags.[http.host/db.connection_string] | -- | +| request_resource | 请求资源 | span.endpointName/span.tags.[http.target/db.statement/messaging.url/rpc.service/cache.key] | If span.tags.http.target exists, read it; if not, extract from http.url, only keeping the call info after the domain name | +| request_id | 请求 ID | -- | | +| response_status | 响应状态 | Try to convert based on response_code; if conversion fails, get span.isError — if true: STATUS_SERVER_ERROR, otherwise STATUS_SERVER_OK | -- | +| response_code | 响应码 | span.tags.[status_code/http.status_code/status.code] | Prefer span.tags.http.status_code | +| response_exception | 响应异常 | Convert to the corresponding exception description based on response_code | -- | +| app_service | 服务名称 | segment.service | -- | +| app_instance | 服务实例 | segment.serviceInstance | -- | +| endpoint | 端点 | span.operationName | -- | +| trace_id | TraceID | span.traceID | -- | +| span_id | SpanID | span.TraceSegmentID-span.spanID | -- | +| parent_span_id | ParentSpanID | segment.ID-span.parentSpanID/span.ref.parentTraceSegmentID-span.ref.parentSpanID | Prefer segment.ID-span.parentSpanID; if span.parentSpanID = -1, then get ParentSpanID from span.ref | +| span_kind | Span 类型 | span.spanType.Exit: SPAN_KIND_CLIENT, span.spanType.Entry: SPAN_KIND_SERVER, span.spanType.Local: SPAN_KIND_INTERNAL, span.spanType.Entry && span.spanLayer.MQ: SPAN_KIND_CONSUMER, span.spanType.Exit && span.spanLayer.MQ: SPAN_KIND_PRODUCER | -- | +| events | 事件 | -- | -- | +| observation_point | 观测点 | span.spanType.Exit: Client Application (C-APP), span.spanType.Entry: Server Application (S-APP), span.spanType.Local: Application (APP) | -- | \ No newline at end of file diff --git a/translate/translated/05-features/02-universal-map/07-metrics-and-operators.md b/translate/translated/05-features/02-universal-map/07-metrics-and-operators.md index e4fbf446..0816c37a 100644 --- a/translate/translated/05-features/02-universal-map/07-metrics-and-operators.md +++ b/translate/translated/05-features/02-universal-map/07-metrics-and-operators.md @@ -1,203 +1,294 @@ --- -title: Metrics and Operators Calculation Logic +title: Calculation Logic for Metrics and Operators permalink: /features/universal-map/metrics-and-operators --- > This document was translated by ChatGPT -This article will introduce different types of metrics and the calculation logic of various operators. +This document introduces the calculation logic for different types of metrics and operators. # Metrics -Metrics are divided into two main categories: `Application Performance Metrics` and `Network Performance Metrics`. +Metrics are divided into two main categories: `application performance metrics` and `network performance metrics`. ## Application Performance Metrics -Application metrics are used to measure the performance of services during actual operation, focusing mainly on service throughput, response delay, and anomalies. By collecting these metrics, operations personnel and developers can better understand the performance of applications in real-world usage, identify potential performance issues, and take appropriate measures for optimization and improvement. +Application metrics are used to measure the performance of services during actual operation, focusing mainly on throughput, response latency, and exceptions. By collecting these metrics, operations and development teams can better understand how applications perform in real-world usage, identify potential performance issues, and take appropriate measures for optimization and improvement. -The metrics described below will record a metric value in each statistical cycle, which can be customized by the user. The system currently supports 1m (one minute) and 1s (one second) by default (these data are collectively referred to as raw data sources in the DeepFlow platform). If multiple metric values are calculated within a statistical cycle, they will be aggregated into one metric value. The aggregation logic is described in the subsequent `Types` section. +The metrics described below record one metric value for each statistical cycle. The statistical cycle can be customized by the user. The system currently supports 1m (one minute) and 1s (one second) by default (these data are collectively referred to as raw data sources in the DeepFlow platform). If multiple metric values are calculated within a single statistical cycle, they will be aggregated into one metric value. The aggregation logic is described later in the `Type` section. ### Throughput -[csv-吞吐量](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/application.en?Category=Throughput) +[csv-Throughput](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/application.en?Category=Throughput) ### Delay -[csv-时延](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/application.en?Category=Delay) +[csv-Delay](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/application.en?Category=Delay) ### Error -[csv-异常](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/application.en?Category=Error) +[csv-Error](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/application.en?Category=Error) ## Network Performance Metrics -Network metrics are quantitative indicators used to evaluate network performance, covering the network layer, transport layer, and application layer. These metrics include throughput, delay, performance, and anomaly types. +Network metrics are quantitative indicators used to evaluate network performance, covering the network layer, transport layer, and application layer. These metrics include throughput, latency, performance, and exception types. ### L3 Throughput -[csv-网络层吞吐](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=L3 Throughput) +[csv-L3 Throughput](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=L3 Throughput) ### L4 Throughput -[csv-传输层吞吐量](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=L4 Throughput) +[csv-L4 Throughput](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=L4 Throughput) Active connection calculation logic: -- The collector counts the raw active connections based on the quadruple (client IP, server IP, protocol, server port) and then calculates the active connections corresponding to resources and paths. -- If traffic is collected within the time interval corresponding to the data source, active connections are counted, but there are some special cases: - - 1s data source: Describes the active connections counted per second. - - The first second of each minute: Includes connections that have no traffic within that second but have not ended, generally used to evaluate concurrent connections (multiple non-overlapping connections with a duration of less than one second may introduce some errors). - - The last 59 seconds of each minute: If multiple flows with the same quadruple have no traffic within that second, the connections corresponding to that quadruple will be ignored for that second, generally used to evaluate the lower bound of concurrent connections. - - 1m data source: Describes the active connections counted per minute. - - Includes connections that have no traffic but have not ended, generally used to evaluate the upper bound of concurrent connections. - - Custom data source: Calculated based on 1s/1m data sources using Avg/Max/Min, with the same meaning as directly using 1s/1m data sources and selecting Avg/Max/Min operators. +- The collector counts the original number of active connections based on the quadruple (client IP, server IP, protocol, server port), and then calculates the active connections corresponding to resources and paths. +- If traffic is captured within the time interval of the data source, active connections are counted, but there are some special cases: + - 1s data source: describes the number of active connections counted per second + - First second of each minute: includes connections without traffic but not yet closed during that second, generally used to estimate concurrent connections (multiple non-overlapping connections lasting less than one second may cause some errors) + - Remaining 59 seconds of each minute: if multiple flows with the same quadruple have no traffic in that second, the connection count for that quadruple is ignored for that second, generally used to estimate the lower bound of concurrent connections + - 1m data source: describes the number of active connections counted per minute + - Includes connections without traffic but not yet closed, generally used to estimate the upper bound of concurrent connections + - Custom data source: calculated from 1s/1m data sources using Avg/Max/Min, with the same meaning as directly using the 1s/1m data source and selecting the Avg/Max/Min operator ### TCP Slow -[csv-传输层 TCP 性能](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=TCP Slow) +[csv-TCP Slow](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=TCP Slow) ### TCP Error -[csv-传输层 TCP 异常](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=TCP Error) - -#### TCP Connection Errors - -![TCP 建连异常](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240411661782557f7bc.png) - -#### TCP Transmission Errors - -![TCP 传输异常](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202404116617825667233.png) - -### Delay - -[csv-传输层时延](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=Delay) - -![TCP 网络时延解剖](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023030364019bdd98b78.jpg) - -- Delay generated during connection establishment - - [1] The complete `connection establishment delay` includes the entire time from the client sending the SYN packet to receiving the SYN+ACK packet from the server and then replying with an ACK packet. The connection establishment delay can be further divided into `client connection establishment delay` and `server connection establishment delay`. - - [2] `Client connection establishment delay` is the time taken for the client to reply with an ACK packet after receiving the SYN+ACK packet. - - [3] `Server connection establishment delay` is the time taken for the server to reply with a SYN+ACK packet after receiving the SYN packet. -- Delay generated during data communication can be divided into `client waiting delay` + `data transmission delay`. - - [4] `Client waiting delay` is the time taken for the client to send the first request after the connection is successfully established; it is also the time taken for the client to send a data packet after receiving a data packet from the server. - - [5] `Data transmission delay` is the time taken for the client to send a data packet and receive a reply data packet from the server. - - [6] During data transmission delay, there is also a delay generated by the system protocol stack, called `system delay`, which is the time taken for the data packet to receive an ACK packet. - -### Application - -[csv-应用层指标](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=Application) +[csv-TCP Error](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=TCP Error) + +#### TCP Client Connection Exceptions + +- Client port reuse + - **Phenomenon**: The server receives SYN but does not reply with SYN-ACK, causing TCP connection failure + - **Cause**: Client source port conflicts with an already established TCP connection + - **Recommendation**: + - Check client TCP connection timeout parameters + - If there is a NAT device, check NAT rules +- Client ACK missing + - **Phenomenon**: The server replies with SYN-ACK, but the client does not respond, causing TCP connection failure + - **Cause**: + - Client SYN Flood attack + - Client port scanning + - **Recommendation**: Confirm whether it is a security incident and block the abnormal client in time +- Other client resets + - **Phenomenon**: The client sends SYN and then immediately sends RST, causing TCP connection failure + - **Cause**: + - Client application exception + - Malicious client attack + - **Recommendation**: + - Check client application status + - Check whether the client has general attack behavior + +![TCP Client Connection Exceptions](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20241014670ce8c1ea0f9.png) + +#### TCP Server Connection Exceptions + +- Server direct reset + - **Phenomenon**: The server receives SYN and replies with RST, rejecting TCP connection + - **Cause**: + - Server port not open or not listening + - Server application not ready + - Client port scanning + - **Recommendation**: + - Check server port connectivity + - Check whether the client is performing port scanning +- Server SYN missing + - **Phenomenon**: The client sends SYN multiple times, but the server does not respond + - **Cause**: + - Firewall not allowing the port + - Route unreachable + - **Recommendation**: + - Check firewall policy + - Check network connectivity +- Other server resets + - **Phenomenon**: The server sends SYN-ACK and then immediately sends RST, causing TCP connection failure + - **Cause**: Server operating system exception + - **Recommendation**: Check server operating system logs + +![TCP Server Connection Exceptions](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20241014670ce89f268c4.png) + +#### TCP Transmission Exceptions + +- Server queue overflow + - **Phenomenon**: During TCP data transmission, the server sends SYN-ACK + - **Cause**: Server Accept queue overflow + - **Recommendation**: + - Adjust kernel somaxconn parameter + - Adjust kernel tcp_max_syn_backlog parameter +- Client reset + - **Phenomenon**: During TCP data transmission, the client sends RST to close the TCP connection + - **Cause**: + - Client application exception + - Client operating system exception + - **Recommendation**: + - Check client application status + - Check client operating system logs +- Server reset + - **Phenomenon**: During TCP data transmission, the server sends RST to close the TCP connection + - **Cause**: + - Server application exception + - Server operating system exception + - **Recommendation**: + - Check server application status + - Check server operating system logs +- TCP connection timeout + - **Phenomenon**: No data for more than 300 seconds during transmission + - **Cause**: + - Client host offline + - Client application exception + - **Recommendation**: + - Check client host status + - Check client application status + +![TCP Transmission Exceptions](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20241014670ce885cccd5.png) + +#### TCP Disconnection Exceptions + +- Server half-close + - **Phenomenon**: The server receives FIN but does not reply with FIN-ACK, resulting in incomplete TCP four-way handshake + - **Cause**: Server application exception + - **Recommendation**: Check server application status +- Client half-close + - **Phenomenon**: The client receives FIN but does not reply with FIN-ACK, resulting in incomplete TCP four-way handshake + - **Cause**: Client application exception + - **Recommendation**: Check client application status + +![TCP Disconnection Exceptions](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20241014670ce893a1ed2.png) + +### Transport Layer Delay + +[csv-Transport Layer Delay](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=Delay) + +![TCP Network Delay Analysis](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023030364019bdd98b78.jpg) + +- Delay during connection establishment + - [1] Complete `connection establishment delay` includes the entire time from when the client sends a SYN packet to receiving the server's SYN+ACK packet and replying with an ACK packet. This can be further divided into `client connection delay` and `server connection delay` + - [2] `Client connection delay` is the time from when the client receives the SYN+ACK packet to when it replies with an ACK packet + - [3] `Server connection delay` is the time from when the server receives the SYN packet to when it replies with a SYN+ACK packet +- Delay during data communication, which can be divided into `client wait delay` + `data transmission delay` + - [4] `Client wait delay` is the time from successful connection establishment to when the client sends the first request; or the time from receiving a data packet from the server to when the client sends another data packet + - [5] `Data transmission delay` is the time from when the client sends a data packet to when it receives the server's reply + - [6] Within data transmission delay, there is also processing delay in the system protocol stack, called `system delay`, which is the time from receiving a data packet to receiving the ACK packet + +### Application Layer Metrics + +[csv-Application Layer Metrics](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=Application) ### Cardinality -During the statistical cycle, the number of unique tags collected is counted. For example, querying the `client IP address (ip_0)` metric for all accesses to pod_1 means counting the number of unique client IP addresses in all traffic accessing pod_1. +Within the statistical cycle, count the number of unique tags in the collected data. For example, querying the metric `client IP address (ip_0)` for all clients accessing pod_1 means counting how many unique client IP addresses appear in all traffic accessing pod_1. -[csv-基数统计](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=Cardinality) +[csv-Cardinality](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/metrics/flow_metrics/network.en?Category=Cardinality) # Operators -Operators calculate data from raw data sources based on the selected time range and interval. For example, using a line chart to view 1s raw data sources for the last 5 minutes with a 20s interval, a point on the line chart (14:43:00) would read all data within the time range of 14:42:40 - 14:43:00 and then calculate the average value. - -Operators support nested stacking, but `aggregate operators` do not support stacking. For example, PerSecond(Avg(byte)) means calculating Avg(byte) first, and then the resulting value is recalculated based on PerSecond. - -## Aggregate Operators - -| Operator | English Name | Applicable Metric Types | Description | -| --------------- | ------------------------------ | ----------------------- | ----------------------------------------------------- | -| Avg | Average | All types | Average value (does not ignore zero values for Counter/Gauge metrics) | -| AAvg | Arithmetic Average | All types | Arithmetic average (first calculate the average at each time point, then calculate the average of the averages) | -| Sum | Sum | Counter type | Sum | -| Max | Maximum | All types | Maximum value | -| Min | Minimum | All types | Minimum value | -| Percentile | Estimated Percentile | All types | Estimated percentile | -| PercentileExact | Exact Percentile | All types | Exact percentile | -| Spread | Spread | All types | Absolute spread, Max minus Min within the statistical cycle | -| Rspread | Relative Spread | All types | Relative spread, Max divided by Min within the statistical cycle | -| Stddev | Standard Deviation | All types | Standard deviation | -| Apdex | Application Performance Index | Delay type | Delay satisfaction | -| Last | Last | All types | Latest value | -| Uniq | Estimated Uniq | Cardinality type | Estimated cardinality | -| UniqExact | Exact Uniq | Cardinality type | Exact cardinality | +Operators calculate data from raw data sources based on the selected time range and interval. For example, when using a line chart to view 1s raw data for the last 5 minutes with a 20s interval and Avg operator, a point at 14:43:00 reads all data from 14:42:40 to 14:43:00 in the raw data source and then calculates the average. + +Operators support nested stacking, but `aggregation operators` do not support stacking. For example, PerSecond(Avg(byte)) means first calculating Avg(byte), then applying PerSecond to the result. + +## Aggregation Operators + +| Operator | English Name | Applicable Metric Type | Description | +| --------------- | ------------------------------ | ---------------------- | --------------------------------------------------------------------------- | +| Avg | Average | All types | Average value (does not ignore zero values for Counter/Gauge metrics) | +| AAvg | Arithmetic Average | All types | Arithmetic average (average of averages at each time point) | +| Sum | Sum | Counter type | Sum | +| Max | Maximum | All types | Maximum value | +| Min | Minimum | All types | Minimum value | +| Percentile | Estimated Percentile | All types | Estimated percentile | +| PercentileExact | Exact Percentile | All types | Exact percentile | +| Spread | Spread | All types | Absolute spread: Max minus Min within the statistical cycle | +| Rspread | Relative Spread | All types | Relative spread: Max divided by Min within the statistical cycle | +| Stddev | Standard Deviation | All types | Standard deviation | +| Apdex | Application Performance Index | Delay type | Delay satisfaction index | +| Last | Last | All types | Latest value | +| Uniq | Estimated Uniq | Cardinality type | Estimated cardinality | +| UniqExact | Exact Uniq | Cardinality type | Exact cardinality | ## Secondary Operators -| Operator | Description | -| ---------- | ------------------------------------------------ | -| PerSecond | Calculate rate, divide the result of the inner operator by the time interval [1] | -| Math | Arithmetic operations, supports +, -, *, / | -| Percentage | Unit conversion % | +| Operator | Description | +| ---------- | ------------------------------------------------------------------------ | +| PerSecond | Calculates rate by dividing the inner operator result by the interval [1]| +| Math | Arithmetic operations: supports +, -, \*, / | +| Percentage | Unit conversion to % | -- [1] For example: `PerSecond(Sum)` means calculating the sum first, then dividing by the time interval `interval` passed by the API; `PerSecond(Avg)` means calculating the average first, then dividing by the data source time interval `data_precision`. +- [1] For example: `PerSecond(Sum)` means summing first, then dividing by the API-provided interval `interval`; `PerSecond(Avg)` means averaging first, then dividing by the data source interval `data_precision`. -# Calculation Logic of Different Metrics' Operators +# Operator Calculation Logic for Different Metrics ## Counter/Gauge Metrics -- flow_metrics data table +- flow_metrics tables - `Sum` operator - - Calculate the `Sum` of all data within the query time range + - Sum all data within the query time range - `Avg` operator - - Calculate the `Sum` of all data within the query time range and divide by `interval/data_precision` + - Sum all data within the query time range, then divide by `interval/data_precision` - Other operators - - First use `Sum` to aggregate based on `data_precision` - - Then call the `ClickHouse` function for the selected specific operator - - When forced (due to the need for other metrics in the same statement) to use two layers of `SQL` calculations + - First aggregate using `Sum` based on `data_precision` + - Then apply the selected operator using `ClickHouse` functions + - When forced (due to other metrics in the same query) to use two-layer SQL calculation - `Sum/Avg` operator - - First use `Sum` to aggregate based on `data_precision` - - Then call the `ClickHouse` function for the selected specific operator -- flow_log data table - - Call the `ClickHouse` function for the selected specific operator -- prometheus/ext_metrics/deepflow_system data table - - Same as flow_metrics data table + - First aggregate using `Sum` based on `data_precision` + - Then apply the selected operator using `ClickHouse` functions +- flow_log tables + - Apply the selected operator using `ClickHouse` functions +- prometheus/ext_metrics/deepflow_system tables + - Same as flow_metrics tables - Additional notes - - The `Min` operator fills 0 for time points with no data or data as `null` + - `Min` operator fills 0 for time points with no data or null values ## Quotient/Percentage Metrics -- flow_metric data table +- flow_metrics tables - `Avg` operator - Calculate `Sum(x)/Sum(y)` for all data within the query time range - Other operators - - First use `Sum(x)/Sum(y)` to aggregate based on `data_precision` - - Then call the `ClickHouse` function for the selected specific operator - - When forced (due to the need for other metrics in the same statement) to use two layers of `SQL` calculations + - First aggregate `Sum(x)/Sum(y)` based on `data_precision` + - Then apply the selected operator using `ClickHouse` functions + - When forced to use two-layer SQL calculation - `Avg` operator - - First use `Sum(x)/Sum(y)` to aggregate based on `data_precision` - - Then call the `ClickHouse` function for the selected specific operator -- flow_log data table - - Call the `ClickHouse` function `func(x/y)` for the selected specific operator + - First aggregate `Sum(x)/Sum(y)` based on `data_precision` + - Then apply the selected operator using `ClickHouse` functions +- flow_log tables + - Apply the selected operator using `ClickHouse` function `func(x/y)` - Additional notes - - The `Min` operator for `Percentage` metrics fills 0 for time points with no data - - When calculating `Sum(x)/Sum(y)`, points with a denominator of `0/null` or a numerator of `null` are ignored + - For `Percentage` metrics, the `Min` operator fills 0 for time points with no data + - When calculating `Sum(x)/Sum(y)`, ignore points where the denominator is `0/null` or the numerator is `null` ## Delay/BoundedGauge Metrics -- flow_metric data table - - Call the `ClickHouse` function for the selected specific operator - - When forced (due to the need for other metrics in the same statement) to use two layers of `SQL` calculations - - `Avg/Min/Max` operator - - Both layers call the `ClickHouse` function for the selected specific operator - - `Spread/Rspread` operator - - First use `Max` and `Min` to aggregate based on `data_precision` - - Then call the `ClickHouse` function for the selected specific operator +- flow_metrics tables + - Apply the selected operator using `ClickHouse` functions + - When forced to use two-layer SQL calculation + - `Avg/Min/Max` operators + - Both layers apply the selected operator using `ClickHouse` functions + - `Spread/Rspread` operators + - First aggregate using `Max` and `Min` based on `data_precision` + - Then apply the selected operator using `ClickHouse` functions - Other operators - - First use `groupArray` to aggregate - - Then call the `ClickHouse` function for the selected specific operator -- flow_log data table - - Call the `ClickHouse` function for the selected specific operator + - First aggregate using `groupArray` + - Then apply the selected operator using `ClickHouse` functions +- flow_log tables + - Apply the selected operator using `ClickHouse` functions - Additional notes - - The `Min` operator for `BoundedGauge` metrics fills 0 for time points with no data or data as `null` - - `Delay` metrics ignore points with a value of 0, considering 0 as a meaningless delay value - -## data_precision of Different Databases/Tables - -| Database | data_precision | Remarks | -| ---------------- | -------------- | -------------------------------------------------------------- | -| flow_metrics | 1s/1m | Supports 1s and 1m by default, can be aggregated to 1h and 1d | -| flow_log | 1s | No actual concept of `data_precision`, the value is for convenience in calculation | -| application_log | 1s | No actual concept of `data_precision`, the value is for convenience in calculation | -| prometheus | 10s | Can be modified through the `data_source_prometheus_interval` field in `server.yaml` | -| ext_metrics | 10s | Can be modified through the `data_source_ext_metrics_interval` field in `server.yaml` | -| deepflow_admin | 10s | | -| deepflow_tenant | 10s | | -| event | 1s | No actual concept of `data_precision`, the value is for convenience in calculation | -| profile | 1s | No actual concept of `data_precision`, the value is for convenience in calculation | + - For `BoundedGauge` metrics, the `Min` operator fills 0 for time points with no data or null values + - For `Delay` metrics, ignore points with value 0, as 0 is considered a meaningless delay value + +## data_precision for Different Databases/Tables + +| Database | data_precision | Notes | +| --------------- | -------------- | --------------------------------------------------------------------- | +| flow_metrics | 1s/1m | Supports 1s and 1m by default, can be aggregated to 1h, 1d | +| flow_log | 1s | No actual `data_precision` concept, value is for calculation purposes | +| application_log | 1s | No actual `data_precision` concept, value is for calculation purposes | +| prometheus | 10s | Can be modified via `data_source_prometheus_interval` in `server.yaml`| +| ext_metrics | 10s | Can be modified via `data_source_ext_metrics_interval` in `server.yaml`| +| deepflow_admin | 10s | | +| deepflow_tenant | 10s | | +| event | 1s | No actual `data_precision` concept, value is for calculation purposes | +| profile | 1s | No actual `data_precision` concept, value is for calculation purposes | \ No newline at end of file diff --git a/translate/translated/05-features/04-continuous-profiling/01-auto-profiling.md b/translate/translated/05-features/04-continuous-profiling/01-auto-profiling.md index f0dcff06..349d0c6e 100644 --- a/translate/translated/05-features/04-continuous-profiling/01-auto-profiling.md +++ b/translate/translated/05-features/04-continuous-profiling/01-auto-profiling.md @@ -7,65 +7,86 @@ permalink: /features/continuous-profiling/auto-profiling # AutoProfiling -By using eBPF to capture snapshots of application function call stacks, DeepFlow can generate Profiling flame graphs for any process, helping developers quickly identify function performance bottlenecks. **In addition to business functions, the function call stack also displays the time consumption of dynamic link libraries, language runtimes, and kernel functions.** Moreover, DeepFlow generates a unique identifier when collecting function call stacks, which can be used to correlate with call logs, enabling the integration of distributed tracing and function performance profiling. +By using eBPF to capture snapshots of an application's function call stack, DeepFlow can generate profiling flame graphs for any process, helping developers quickly pinpoint function performance bottlenecks. **In addition to business functions, the function call stack can also display the time consumption of dynamic link libraries, language runtimes, and kernel functions**. Furthermore, when collecting function call stacks, DeepFlow generates a unique identifier that can be associated with call logs, enabling the linkage between distributed tracing and function performance profiling. ![DeepFlow's AutoProfiling](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240601665a96f4b63fd.png) # Capabilities and Limitations -Supported eBPF Profiling data types: +Supported eBPF profiling data types: | Type | Supported Languages/Libraries | Community Edition | Enterprise Edition | | --------- | ----------------------------- | ----------------- | ------------------ | -| on-cpu | Java | ✔ | ✔ | -| | C/C++ | ✔ | ✔ | -| | Rust | ✔ | ✔ | -| | Golang | ✔ | ✔ | -| | Python `*` | ✔ | ✔ | -| | CUDA `*` | ✔ | ✔ | -| | Lua `*` | ✔ | ✔ | -| off-cpu | Java | | ✔ | -| | C/C++ | | ✔ | -| | Rust | | ✔ | -| | Golang | | ✔ | -| | Python `*` | | ✔ | -| | CUDA `*` | | ✔ | -| | Lua `*` | | ✔ | -| mem-alloc | Java | | ✔ | -| | Rust `*` | | ✔ | -| | Golang `*` | | ✔ | -| | Python `*` | | ✔ | -| mem-inuse | Rust `*` | | ✔ | -| hbm-alloc | Python `*` | | ✔ | -| hbm-inuse | Python `*` | | ✔ | -| rdma | C/C++ `*` | | ✔ | +| on-cpu | Java | ✔ | ✔ | +| | C/C++ | ✔ | ✔ | +| | Rust | ✔ | ✔ | +| | Golang | ✔ | ✔ | +| | Python `***` | ✔ | ✔ | +| | CUDA | ✔ | ✔ | +| | Lua `*` | ✔ | ✔ | +| off-cpu | Java | | ✔ | +| | C/C++ | | ✔ | +| | Rust | | ✔ | +| | Golang | | ✔ | +| | Python `***` | | ✔ | +| | CUDA | | ✔ | +| | Lua `*` | | ✔ | +| on-gpu | CUDA `*` | | ✔ | +| mem-alloc | Java `**` | | ✔ | +| | Rust | | ✔ | +| | Golang `*` | | ✔ | +| | Python `*` `***` | | ✔ | +| mem-inuse | Rust | | ✔ | +| hbm-alloc | CUDA `*` | | ✔ | +| hbm-inuse | CUDA `*` | | ✔ | +| rdma | C/C++ `*` | | ✔ | Notes: -- `*`: features in development -- Types: - - on-cpu: Time spent by functions on the CPU - - off-cpu: Time functions wait for the CPU - - mem-alloc: Total memory allocation of objects and function call stacks - - mem-inuse: Current memory usage of objects and function call stacks - - hbm-alloc: Total GPU memory allocation of objects and function call stacks - - hbm-inuse: Current GPU memory usage of objects and function call stacks -- Languages: - - Languages compiled into ELF format executables: Golang, Rust, C/C++ - - Languages using the JVM: Java - -Two prerequisites must be met to obtain Profiling data: - -- The process needs to enable Frame Pointer - - Compiling C/C++: `gcc -fno-omit-frame-pointer` - - Compiling Rust: `RUSTFLAGS="-C force-frame-pointers=yes"` - - Compiling Golang: Enabled by default, no additional compilation parameters needed - - Running Java: `-XX:+PreserveFramePointer` -- For processes of compiled languages, ensure to retain the symbol table during compilation - -The Off-CPU Profiling feature **only** collects the following call stacks: - -- Call stacks where the process state **equals** `TASK_INTERRUPTIBLE` (interruptible sleep) or `TASK_UNINTERRUPTIBLE` (uninterruptible sleep) when yielding the CPU -- Call stacks **excluding** the 0th process (Idle process) -- Call stacks containing **at least one** user-mode function -- Call stacks where the CPU wait time **does not exceed** 1 hour \ No newline at end of file +- `*`: features in development +- `**`: The JVM running the Java program must have a symbol table, see [check method](#jvm-symbol-table-check) +- `***`: Currently supports Python 3.10 +- Types: + - on-cpu: Time a function spends on the CPU + - off-cpu: Time a function waits for the CPU + - on-gpu: Time a function spends on the GPU + - mem-alloc: Total memory allocated by objects and the function call stack + - mem-inuse: Current memory usage of objects and the function call stack + - hbm-alloc: Total GPU memory allocated by objects and the function call stack + - hbm-inuse: Current GPU memory usage of objects and the function call stack +- Languages: + - Languages compiled into ELF format executables: Golang, Rust, C/C++ + - Languages using the JVM: Java + - Interpreted languages: Python + +Two prerequisites must be met to obtain profiling data: + +- The application process must enable Frame Pointer or enable the Agent's DWARF stack unwinding capability + - Enable Frame Pointer (frame pointer register) for the application process: + - Compile C/C++: `gcc -fno-omit-frame-pointer` + - Compile Rust: `RUSTFLAGS="-C force-frame-pointers=yes"` + - Compile Golang: Enabled by default, no extra compile parameters needed + - Run Java: `-XX:+PreserveFramePointer` + - For enabling the Agent's DWARF stack unwinding capability, please refer to the [documentation](../../configuration/agent/#inputs.ebpf.profile.unwinding) +- For compiled languages, ensure the symbol table is preserved during compilation + +The Off-CPU profiling feature **only** collects the following call stacks: + +- Call stacks where the process state is **equal to** `TASK_INTERRUPTIBLE` (interruptible sleep) or `TASK_UNINTERRUPTIBLE` (uninterruptible sleep) when yielding the CPU +- Call stacks **excluding** process 0 (Idle process) +- Call stacks containing **at least one** user-space function +- Call stacks where the CPU wait time is **no more than** 1 hour + +# FAQ + +## JVM Symbol Table Check + +- Find the process ID of the Java process that requires memory profiling, denoted as `$pid` +- Check the location of the loaded `libjvm.so` for the process, denoted as `$path` + ``` + grep libjvm.so /proc/$pid/maps + ``` +- Check whether the file contains a symbol table + ``` + readelf -WS $path | grep symtab + ``` \ No newline at end of file diff --git a/translate/translated/05-features/05-auto-tagging/04-custom-tags.md b/translate/translated/05-features/05-auto-tagging/04-custom-tags.md index ca638279..125cfce6 100644 --- a/translate/translated/05-features/05-auto-tagging/04-custom-tags.md +++ b/translate/translated/05-features/05-auto-tagging/04-custom-tags.md @@ -7,10 +7,10 @@ permalink: /features/auto-tagging/custom-tags # K8s Label -DeepFlow currently supports automatically associating K8s custom Labels with the following resources: +DeepFlow currently supports automatically associating K8s custom Label resources, including: -- Container services -- Workloads +- Container Service +- Workload - Deployment - StatefulSet - DaemonSet @@ -22,22 +22,22 @@ DeepFlow currently supports automatically associating K8s custom Labels with the # K8s Annotation -DeepFlow (Enterprise Edition only) currently supports automatically associating K8s custom Annotations with the following resources: +DeepFlow (Enterprise Edition only) currently supports automatically associating K8s custom Annotation resources, including: -- Container services +- Container Service - Pod # K8s Env -DeepFlow (Enterprise Edition only) currently supports automatically associating K8s custom Annotations with the following resources: +DeepFlow (Enterprise Edition only) currently supports automatically associating K8s custom Annotation resources, including: - Pod # Cloud Resource Custom Tags -DeepFlow (Enterprise Edition only) currently supports automatically associating cloud resource custom tags with the following resources: +DeepFlow (Enterprise Edition only) currently supports automatically associating cloud resource custom tags for the following resources: -- Cloud servers +- Cloud Server Supported public cloud providers include: @@ -47,40 +47,40 @@ Supported public cloud providers include: Supported private cloud providers include: -- Aliyun Dedicated Cloud +- Alibaba Private Cloud # Custom Auto Grouping Tags -The DeepFlow system provides two default auto-grouping tags: auto_instance and auto_service: +The DeepFlow system provides two default auto-grouping tags: `auto_instance` and `auto_service`: -- auto_instance: Automatically identifies the corresponding instance tag based on IP or process ID. The system sets recognizable tags and their priorities (Container POD > Process > Container Node > Others > IP). +- auto_instance: Automatically identifies the corresponding instance tag based on IP or process ID. The system has predefined recognizable tags and their priorities (Container POD > Process > Container Node > Others > IP) [csv-auto_instance_type](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/tag/enum/auto_instance_type.en) -- auto_service: Automatically identifies the corresponding service tag based on IP or process ID. The system sets recognizable tags and their priorities (Container Service > Workload > Process > Container Cluster > Others > IP). - - Compared to auto_instance, auto_service removes `Container POD` and adds `Container Service` and `Workload` tags that better reflect services. The `Container Service` tag has a higher priority than `Workload`, so when an IP belongs to both `Container Service` and `Workload`, it will be identified as `Container Service`. +- auto_service: Automatically identifies the corresponding service tag based on IP or process ID. The system has predefined recognizable tags and their priorities (Custom Service > Container Service > Workload > Process > Container Cluster > Others > IP) + - Compared to `auto_instance`, `auto_service` removes `Container POD` and adds `Custom Service` (Enterprise Edition only), `Container Service`, and `Workload` tags that better represent services. Among them, `Container Service` has a higher recognition priority than `Workload`. Therefore, when an IP belongs to both `Container Service` and `Workload`, it will be recognized as `Container Service`. [csv-auto_service_type](https://raw.githubusercontent.com/deepflowio/deepflow/main/server/querier/db_descriptions/clickhouse/tag/enum/auto_service_type.en) -DeepFlow also supports the ability to customize auto-grouping tags. You can configure the tags and their priorities as needed. The configuration document is as follows: +DeepFlow also supports defining custom auto-grouping tags. You can configure the tags to be recognized and their priorities as needed. The configuration document is as follows: ```yaml querier: - auto-custom-tag: + auto-custom-tags: # The Name of Custom Tag # Note: Cannot use colon, space, or backquote. - tag-name: auto_my_tag - # The Value of Custom Tag - # Note: Range of source tags for retrieving the field value. Each row of data will - # automatically use the first non-zero tag encountered from top to bottom as the - # value for the custom tag. Here you can enter any tags seen in the results of - # the `show tags from
` API. - tag-values: - - k8s.label.app - - auto_service + - tag-name: auto_my_tag + # The Value of Custom Tag + # Note: Range of source tags for retrieving the field value. Each row of data will + # automatically use the first non-zero tag encountered from top to bottom as the + # value for the custom tag. Here you can enter any tags seen in the results of + # the `show tags from
` API. + tag-fields: + - k8s.label.app + - auto_service ``` -The `$tag-name` tag defined above is used similarly to `auto_instance` and `auto_service`, with the following additional restrictions: +The `$tag-name` tag defined above works basically the same as `auto_instance` and `auto_service`, with the following additional restrictions: -- The `$tag-name` defined here cannot be used for grouping with `*` simultaneously. -- The `$tag-name` defined here cannot be used for grouping with any tags included in the `$tag-values` definition simultaneously. \ No newline at end of file +- The `$tag-name` defined here cannot be used for grouping together with `*` +- The `$tag-name` defined here cannot be used for grouping together with any tags included in `$tag-fields` in its definition \ No newline at end of file diff --git a/translate/translated/05-features/05-auto-tagging/06-additional-cloud-tags.md b/translate/translated/05-features/05-auto-tagging/06-additional-cloud-tags.md index af1282bc..58b97c70 100644 --- a/translate/translated/05-features/05-auto-tagging/06-additional-cloud-tags.md +++ b/translate/translated/05-features/05-auto-tagging/06-additional-cloud-tags.md @@ -7,7 +7,8 @@ permalink: /features/auto-tagging/additional-cloud-tags # Introduction -In addition to actively calling (pulling) APIs from cloud service providers and K8s apiserver to synchronize resource information, DeepFlow also provides a declarative interface `domain-additional-resource` to allow external services to push additional resource information. This method is suitable for synchronizing public cloud resources not yet supported by DeepFlow, synchronizing private cloud resources using the DeepFlow community edition, and synchronizing business tags in CMDB. +In addition to actively invoking (pulling) the APIs of cloud service providers and the K8s apiserver to synchronize resource information, DeepFlow also provides a declarative interface `domain-additional-resource` that allows external services to push additional resource information. +This method is applicable for synchronizing public cloud resources not yet supported by DeepFlow, synchronizing private cloud resources using the DeepFlow community edition, and synchronizing business tags from CMDB, among other scenarios. The resource information that can be pushed using this API includes: @@ -21,12 +22,12 @@ The resource information that can be pushed using this API includes: The custom tags that can be pushed using this API include: -- Custom tags associated with the following K8s resources +- Custom tags associated with the following K8s resources: - Namespace -- Custom tags associated with the following cloud resources +- Custom tags associated with the following cloud resources: - Cloud Server -# API Call Method +# API Usage ## Endpoint @@ -48,125 +49,125 @@ PUT ### Body -| Name | Type | Required | Description | -| ---------- | ------------------- | -------- | ------------------------------------------------------------------- | -| azs | Array of AZ Structs | No | Availability Zone | -| vpcs | Array of VPC Structs| No | Virtual Private Cloud | -| subnets | Array of Subnet Structs | No | Subnet | -| hosts | Array of Host Structs | No | Server | -| chosts | Array of Chost Structs | No | Cloud Server | -| cloud_tags | Array of CloudTag Structs | No | Generally used to inject business tags, see [Business Tags in CMDB](./cmdb-tags/) | -| lbs | Array of LB Structs | No | Load Balancer | - -AZ Struct +| Name | Type | Required | Description | +| ---------- | --------------------- | -------- | --------------------------------------------------------------------------- | +| azs | AZ struct array | No | Availability Zone | +| vpcs | VPC struct array | No | Virtual Private Cloud | +| subnets | Subnet struct array | No | Subnet | +| hosts | Host struct array | No | Server | +| chosts | Chost struct array | No | Cloud Server | +| cloud_tags | CloudTag struct array | No | Generally used to inject business tags, see [Business Tags in CMDB](./cmdb-tags/) | +| lbs | LB struct array | No | Load Balancer | + +AZ struct | Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | +| ---------- | ------ | -------- | ----------------- | | name | String | Yes | | | uuid | String | Yes | | -| domain_uuid| String | Yes | Cloud platform UUID | +| domain_uuid| String | Yes | Cloud platform UUID| -VPC Struct +VPC struct | Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | +| ---------- | ------ | -------- | ----------------- | | name | String | Yes | | | uuid | String | Yes | | -| domain_uuid| String | Yes | Cloud platform UUID | - -Subnet Struct -| Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | -| name | String | Yes | | -| uuid | String | Yes | | -| type | Integer| No | Default: 4, Options: 3 (WAN), 4 (LAN) | -| is_vip | Boolean| No | Options: true, false | -| vpc_uuid | String | Yes | | -| az_uuid | String | No | | -| domain_uuid| String | Yes | Cloud platform UUID | -| cidrs | Array of Strings | Yes | Example: ["x.x.x.x/x"] | - -Host Struct -| Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | -| name | String | Yes | | -| uuid | String | Yes | | -| ip | String | Yes | | -| type | Integer| No | Default: 3 (KVM). Options: 2 (ESXi), 3 (KVM), 5 (Hyper-V), 6 (Gateway) | -| az_uuid | String | Yes | | -| domain_uuid| String | Yes | Cloud platform UUID | -| vinterfaces| Array of Vinterface 1 Structs | No | Network interfaces | - -Vinterface 1 Struct -| Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | -| mac | String | Yes | Example: xx:xx:xx:xx:xx:xx | -| subnet_uuid| String | Yes | | -| ips | Array of Strings | No | Example: ["x.x.x.x"] | - -Chost Struct -| Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | -| name | String | Yes | | -| uuid | String | Yes | | -| host_ip | String | No | Hypervisor IP address | -| type | Integer| No | Default: 1 (vm/compute). Options: 1 (vm/compute), 2 (bm/compute), 3 (vm/network), 4 (bm/network), 5 (vm/storage), 6 (bm/storage) | -| vpc_uuid | String | Yes | | -| az_uuid | String | Yes | | -| domain_uuid| String | Yes | Cloud platform UUID | -| vinterfaces| Array of Vinterface 2 Structs | No | Chost interfaces | - -Vinterface 2 Struct -| Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | -| mac | String | Yes | Example: xx:xx:xx:xx:xx:xx | -| subnet_uuid| String | Yes | | -| ips | Array of Strings | Yes | Example: ["x.x.x.x"] | - -CloudTag Struct: See [Business Tags in CMDB](./cmdb-tags/). - -Tag Struct -| Name | Type | Required | Description | -| ----- | ------ | -------- | ----------------- | -| key | String | Yes | Limit 255 characters, no spaces, colons, backticks, backslashes, single quotes | -| value | String | Yes | Limit 255 characters, no spaces, colons, backticks, backslashes | - -LB Struct -| Name | Type | Required | Description | -| ------------ | ------ | -------- | ----------------- | -| name | String | Yes | | -| model | Integer| Yes | Default: 2. Options: 1 (internal), 2 (external) | -| vpc_uuid | String | Yes | | -| domain_uuid | String | Yes | | -| vinterfaces | Array of Vinterface 2 Structs | No | Chost interfaces | -| lb_listeners | Array of LBListener Structs | No | | - -LBListener Struct -| Name | Type | Required | Description | -| -----------| ------ | -------- | ----------------- | -| name | String | No | If empty, assigned as ${ip}-${port} | -| protocol | String | Yes | Options: TCP, UDP | -| ip | Integer| Yes | | -| port | String | Yes | | -| lb_target_servers | Array of LBTargetServer Structs | No | | - -LBTargetServer Struct -| Name | Type | Required | Description | -| ---- | ------ | -------- | ----------------- | -| ip | String | Yes | | -| port | Integer| Yes | | - -## Response Results +| domain_uuid| String | Yes | Cloud platform UUID| + +Subnet struct +| Name | Type | Required | Description | +| ---------- | -------- | -------- | ------------------------------------------------ | +| name | String | Yes | | +| uuid | String | Yes | | +| type | Integer | No | Default: 4, Options: 3 (WAN), 4 (LAN) | +| is_vip | Boolean | No | Options: true, false | +| vpc_uuid | String | Yes | | +| az_uuid | String | No | | +| domain_uuid| String | Yes | Cloud platform UUID | +| cidrs | String[] | Yes | Example: ["x.x.x.x/x"] | + +Host struct +| Name | Type | Required | Description | +| ---------- | --------------------- | -------- | --------------------------------------------------------------------------- | +| name | String | Yes | | +| uuid | String | Yes | | +| ip | String | Yes | | +| type | Integer | No | Default: 3 (KVM). Options: 2 (ESXi), 3 (KVM), 5 (Hyper-V), 6 (Gateway) | +| az_uuid | String | Yes | | +| domain_uuid| String | Yes | Cloud platform UUID | +| vinterfaces| Vinterface 1 struct[] | No | Network interfaces | + +Vinterface 1 struct +| Name | Type | Required | Description | +| ---------- | -------- | -------- | ---------------------------- | +| mac | String | Yes | Example: xx:xx:xx:xx:xx:xx | +| subnet_uuid| String | Yes | | +| ips | String[] | No | Example: ["x.x.x.x"] | + +Chost struct +| Name | Type | Required | Description | +| ---------- | --------------------- | -------- | --------------------------------------------------------------------------- | +| name | String | Yes | | +| uuid | String | Yes | | +| host_ip | String | No | Hypervisor IP address | +| type | Integer | No | Default: 1 (vm/compute). Options: 1 (vm/compute), 2 (bm/compute), 3 (vm/network), 4 (bm/network), 5 (vm/storage), 6 (bm/storage) | +| vpc_uuid | String | Yes | | +| az_uuid | String | Yes | | +| domain_uuid| String | Yes | Cloud platform UUID | +| vinterfaces| Vinterface 2 struct[] | No | Chost interfaces | + +Vinterface 2 struct +| Name | Type | Required | Description | +| ---------- | -------- | -------- | ---------------------------- | +| mac | String | Yes | Example: xx:xx:xx:xx:xx:xx | +| subnet_uuid| String | Yes | | +| ips | String[] | Yes | Example: ["x.x.x.x"] | + +CloudTag struct: See [Business Tags in CMDB](./cmdb-tags/). + +Tag struct +| Name | Type | Required | Description | +| ----- | ------ | -------- | --------------------------------------------------------------------------- | +| key | String | Yes | Limit 255 characters, no spaces, colons, backticks, backslashes, or quotes | +| value | String | Yes | Limit 255 characters, no spaces, colons, backticks, backslashes | + +LB struct +| Name | Type | Required | Description | +| ---------- | --------------------- | -------- | ------------------------------------------------ | +| name | String | Yes | | +| model | Integer | Yes | Default: 2. Options: 1 (internal), 2 (external) | +| vpc_uuid | String | Yes | | +| domain_uuid| String | Yes | | +| vinterfaces| Vinterface 2 struct[] | No | Chost interfaces | +| lb_listeners| LBListener struct[] | No | | + +LBListener struct +| Name | Type | Required | Description | +| --------------- | ----------------------- | -------- | ---------------------------------------- | +| name | String | No | If empty, defaults to ${ip}-${port} | +| protocol | String | Yes | Options: TCP, UDP | +| ip | Integer | Yes | | +| port | String | Yes | | +| lb_target_servers| LBTargetServer struct[]| No | | + +LBTargetServer struct +| Name | Type | Required | Description | +| ---- | ------ | -------- | ----------- | +| ip | String | Yes | | +| port | Int | Yes | | + +## Response ### Return Parameters | Name | Type | Required | Description | | ----------- | ------ | -------- | ----------------- | -| OPT_STATUS | String | Yes | Success or failure | -| DESCRIPTION | String | Yes | Error information | -| DATA | JSON | Yes | Return data | +| OPT_STATUS | String | Yes | Success or failure| +| DESCRIPTION | String | Yes | Error message | +| DATA | JSON | Yes | Returned data | -### Successful Response +### Success Response -When the return parameter OPT_STATUS equals SUCCESS, it indicates a successful call. Example return value: +When `OPT_STATUS` equals `SUCCESS`, the call is successful. Example: ```json { @@ -176,9 +177,9 @@ When the return parameter OPT_STATUS equals SUCCESS, it indicates a successful c } ``` -### Failed Response +### Failure Response -When the return parameter OPT_STATUS does not equal SUCCESS, it indicates a failed call. Example return value: +When `OPT_STATUS` is not `SUCCESS`, the call failed. Example: ```json { @@ -188,15 +189,15 @@ When the return parameter OPT_STATUS does not equal SUCCESS, it indicates a fail } ``` -The error code is the information in the OPT_STATUS field of the return value -| Error Code | Description | Suggested Solution | -| ------------------ | ----------- | ------------------ | -| INVALID_POST_DATA | Invalid parameter | Check if the corresponding field values are correct based on the error message | -| RESOURCE_NOT_FOUND | Resource not found | The resource value filled in is invalid, please check and fill in correctly | +Error codes are indicated in the `OPT_STATUS` field: +| Error Code | Description | Suggested Solution | +| ------------------ | ----------------- | ------------------------------------------------------- | +| INVALID_POST_DATA | Invalid parameter | Check whether the value of the corresponding field is correct according to the error message | +| RESOURCE_NOT_FOUND | Resource not found| The resource value is invalid, please check and correct | -# Call Example +# Example Usage -## Calling via HTTP API +## Call via HTTP API ```bash curl -XPUT -H "Content-Type:application/json" \ @@ -204,7 +205,7 @@ ${deepflow_server_node_ip}:${port}/v1/domain-additional-resources/ \ -d@additional_resource.json ``` -Parameter file additional_resource.json ([Reference YAML file](https://github.com/deepflowio/deepflow/blob/main/cli/ctl/example/domain_additional_resource.yaml)) +Parameter file `additional_resource.json` ([Reference YAML file](https://github.com/deepflowio/deepflow/blob/main/server/controller/model/domain_additional_resource_example.yaml)) ```json { @@ -315,9 +316,9 @@ Parameter file additional_resource.json ([Reference YAML file](https://github.co } ``` -## Calling via deepflow-ctl Command +## Call via `deepflow-ctl` Command -In addition to using the HTTP API, you can also use the deepflow-ctl command to call via a YAML file. +In addition to using the HTTP API, you can also use the `deepflow-ctl` command with a YAML file. ```bash # View YAML parameter example @@ -330,12 +331,12 @@ deepflow-ctl domain additional-resource example > additional-resource.yaml deepflow-ctl domain additional-resource apply -f additional-resource.yaml ``` -After the resource is manually added successfully, the corresponding database table can be viewed after 1 minute (depending on the resource_recorder_interval field in the server.yaml configuration file): +Once the resource is manually added successfully, after 1 minute (depending on the `resource_recorder_interval` field in the `server.yaml` configuration file), the corresponding database tables will display the information: -- Availability Zone (table az) -- VPC (table epc) -- Subnet (table subnet) -- Server (table host_device) -- Cloud Server (table vm) -- Load Balancer (tables lb, lb_listener, and lb_target_server) -- Namespace (table pod_namespace) \ No newline at end of file +- Availability Zone (table `az`) +- VPC (table `epc`) +- Subnet (table `subnet`) +- Server (table `host_device`) +- Cloud Server (table `vm`) +- Load Balancer (tables `lb`, `lb_listener`, and `lb_target_server`) +- Namespace (table `pod_namespace`) \ No newline at end of file diff --git a/translate/translated/05-features/05-auto-tagging/08-k8s-crd.md b/translate/translated/05-features/05-auto-tagging/08-k8s-crd.md index 2370bfaa..a0d6d6c2 100644 --- a/translate/translated/05-features/05-auto-tagging/08-k8s-crd.md +++ b/translate/translated/05-features/05-auto-tagging/08-k8s-crd.md @@ -1,36 +1,46 @@ --- -title: K8s CRD Labels +title: K8s Custom Resource Tagging permalink: /features/auto-tagging/k8s-crd --- > This document was translated by ChatGPT -# Common special K8s resources or CRDs +# Common Special K8s Resources or CRDs -When an unsynchronized (workload-unassociated) container Pod is detected: -- If the value of Pod's `metadata.ownerReferences[].apiVersion = apps.kruise.io/v1beta1`, then the corresponding K8s platform should be OpenKruise. -- If the value of Pod's `metadata.ownerReferences[].apiVersion = opengauss.sig/v1`, then the corresponding K8s platform should be OpenGauss. +When an unsynchronized container Pod (one without a corresponding workload) is detected: -In these scenarios, the following operations are required: +- If the Pod's `metadata.ownerReferences[].apiVersion = apps.kruise.io/v1beta1`, then the corresponding K8s platform should be OpenKruise. +- If the Pod's `metadata.ownerReferences[].apiVersion = opengauss.sig/v1`, then the corresponding K8s platform should be OpenGauss. -- Enable and disable the corresponding resources in the Agent advanced configuration -- Configure Kubernetes API permissions +In such cases, the following actions are required: + +- Enable or disable the corresponding resources in the Agent configuration. +- Configure Kubernetes API permissions in the Agent's deployment cluster. ## OpenShift -In this scenario, the default `Ingress` resource acquisition needs to be disabled, and the `Route` resource acquisition needs to be enabled. +In this scenario, you need to disable the default `Ingress` resource retrieval and enable `Route` resource retrieval. + +- [Route](https://docs.redhat.com/en/documentation/openshift_container_platform/4.14/html/network_apis/route-route-openshift-io-v1) + + ```yaml + apiVersion: route.openshift.io/v1 + kind: Route + ``` -Agent advanced configuration is as follows: +Modify the Agent configuration as follows: ```yaml -static_config: - kubernetes-resources: - - name: ingresses - disabled: true - - name: routes +inputs: + resources: + kubernetes: + api_resources: + - name: ingresses + disabled: true + - name: routes ``` -ClusterRole configuration addition: +In the container cluster where the Agent is deployed, modify the Agent's ClusterRole configuration to add the following rules: ```yaml rules: @@ -46,24 +56,42 @@ rules: ## OpenKruise -In this scenario, the `CloneSet` and `apps.kruise.io/StatefulSet` resources need to be obtained from the API. +In this scenario, you need to retrieve `CloneSet` and `Advanced StatefulSet` resources from the API. + +- [CloneSet](https://openkruise.io/docs/user-manuals/cloneset/) + + ```yaml + apiVersion: apps.kruise.io/v1alpha1 + kind: CloneSet + ``` -Agent advanced configuration is as follows: +- [Advanced StatefulSet](https://openkruise.io/docs/user-manuals/advancedstatefulset/) + + ```yaml + apiVersion: apps.kruise.io/v1beta1 + kind: StatefulSet + ``` + +Modify the Agent configuration as follows: ```yaml -static_config: - kubernetes-resources: - - name: clonesets - group: apps.kruise.io - - name: statefulsets - group: apps - - name: statefulsets - group: apps.kruise.io +inputs: + resources: + kubernetes: + api_resources: + - name: clonesets + group: apps.kruise.io + - name: statefulsets + group: apps + - name: statefulsets + group: apps.kruise.io ``` -Note that Kubernetes's `apps/StatefulSet` needs to be added here. +::: tip +Since `statefulsets` has the same name in both the `apps` and `apps.kruise.io` groups, if you need to retrieve Kubernetes `StatefulSet` as well, you must enable resource synchronization for both `group=apps.kruise.io, name=statefulsets` and `group=apps, name=statefulsets`. +::: -ClusterRole configuration addition: +In the container cluster where the Agent is deployed, modify the Agent's ClusterRole configuration to add the following rules: ```yaml - apiGroups: @@ -79,17 +107,21 @@ ClusterRole configuration addition: ## OpenGauss -In this scenario, the `OpenGaussCluster` resource needs to be obtained from the API. +In this scenario, you need to retrieve the `OpenGaussCluster` resource from the API. -Agent advanced configuration is as follows: +- [OpenGaussCluster](https://github.com/opengauss-mirror/openGauss-operator) + +Modify the Agent configuration as follows: ```yaml -static_config: - kubernetes-resources: - - name: opengaussclusters +inputs: + resources: + kubernetes: + api_resources: + - name: opengaussclusters ``` -ClusterRole configuration addition: +In the container cluster where the Agent is deployed, modify the Agent's ClusterRole configuration to add the following rules: ```yaml - apiGroups: @@ -102,22 +134,23 @@ ClusterRole configuration addition: - watch ``` -# Other K8s CRD -## About Lua Plugin on the Server +# Other K8s Custom Resources + +## About Server-Side Lua Plugins -Due to some users' Kubernetes environments possibly having special configurations or security requirements, the standardized way of extracting workload types and workload names may not work as expected. Alternatively, users might want to customize workload types and workload names based on their own logic. Therefore, DeepFlow supports users in extracting workload types and workload names by adding custom Lua plugins. The Lua plugin system enhances the flexibility and universality of K8s resource integration by calling Lua Functions at fixed points to obtain some user-defined workload types and names. +Some users' Kubernetes environments may have special configurations or security requirements that prevent the standardized extraction of workload type and workload name from working as expected, or users may want to customize workload type and name extraction based on their own logic. To address this, DeepFlow allows users to extract workload type and name by adding custom Lua plugins. The Lua plugin system calls a Lua function at fixed points to obtain user-defined workload types and names, improving the flexibility and universality of K8s resource integration. -## Lua Plugin Writing Example +## Example of Writing a Lua Plugin ```lua -- Fixed syntax to import the JSON parsing package package.path = package.path..";/bin/?.lua" local dkjson = require("dkjson") --- Fixed function name and parameters +-- Function name and parameters are fixed function GetWorkloadTypeAndName(metaJsonStr) -- Example of metadata JSON - -- Note that the JSON passed in after the colon on this line "metadata": { + -- Note: The colon here is followed by the incoming JSON "metadata": { -- "annotations": { -- "checksum/config": "", -- "cni.projectcalico.org/containerID": "", @@ -149,33 +182,33 @@ function GetWorkloadTypeAndName(metaJsonStr) local metaData = dkjson.decode(metaJsonStr,1,nil) local workloadType = "" local workloadName = "" - -- Note that for customization flexibility, each pod metadata JSON string will be passed to the Lua script, and you need to filter out pods that do not require customization - -- For those that meet the filter criteria, return two empty strings - if condition then -- Here, condition is the criteria you need to filter pods + -- For flexibility, each pod's metadata JSON string will be passed to the Lua script, so you need to filter out pods that don't require customization + -- For those that match the filter condition, return two empty strings + if condition then -- Replace 'condition' with your pod filtering logic return "", "" -- Return two empty strings end - -- Return workloadType and workloadName through custom analysis of metaData + -- Perform custom analysis on metaData and return workloadType and workloadName return workloadType, workloadName end ``` -## Upload Plugin +## Uploading the Plugin -Lua plugins support runtime loading. After uploading the plugin using the deepflow-ctl tool in the DeepFlow runtime environment, it will be automatically loaded. Execute the following command in the environment: +Lua plugins support runtime loading. After uploading the plugin using the `deepflow-ctl` tool in the DeepFlow runtime environment, it will be automatically loaded. Run the following command in the environment: ```sh -# Replace /home/tom/hello.lua with the path to your Lua plugin and hello with the name you want to give the plugin +# Replace /home/tom/hello.lua with the path to your Lua plugin, and hello with the desired plugin name deepflow-ctl plugin create --type lua --image /home/tom/hello.lua --name hello --user server ``` -DeepFlow supports loading multiple Lua plugins simultaneously. If you want different plugins to act on different Pods, make sure to write the filter rules properly. You can view the names of the plugins you have loaded with the following command: +DeepFlow supports loading multiple Lua plugins simultaneously. If you want different plugins to apply to different Pods, make sure to write appropriate filtering rules. You can view the names of loaded plugins with: ```sh deepflow-ctl plugin list ``` -You can delete a specific plugin by its name with the following command: +You can delete a plugin by its name with the following command: ```sh deepflow-ctl plugin delete @@ -183,7 +216,7 @@ deepflow-ctl plugin delete ## Example -For instance, if the metadata of a Pod in the current k8s environment is as follows: +For example, if the metadata of a Pod in the k8s environment is as follows: ```json "metadata": { @@ -207,7 +240,7 @@ For instance, if the metadata of a Pod in the current k8s environment is as foll } ``` -The standardized way cannot extract the workload type and workload name because the kind of data is `OpenGaussCluster`, which is not a type supported by DeepFlow. The currently supported workload types are: Deployment/StatefulSet/DaemonSet/CloneSet. You can write the following Lua script to convert the workload type to a type supported by DeepFlow: +Using the standardized method, the workload type and workload name cannot be extracted because the `kind` is `OpenGaussCluster`, which is not a type supported by DeepFlow. The currently supported workload types are: Deployment / StatefulSet / DaemonSet / CloneSet. You can write the following Lua script to convert the workload type into one supported by DeepFlow: ```lua package.path = package.path..";/bin/?.lua" @@ -219,22 +252,22 @@ function GetWorkloadTypeAndName(metaJsonStr) local meteTable = ownerReferencesData[1] or {} local workloadType = "" local workloadName = "" - -- Get workloadType and ensure it is of string type + -- Get workloadType and ensure it is a string workloadType = tostring(meteTable["kind"] or "") - -- If we only want to process workloadType = "OpenGaussCluster", we can filter out other Pod metadata here + -- If we only want to process workloadType = "OpenGaussCluster", filter out other Pods here if workloadType ~= "OpenGaussCluster" then - -- Directly return empty strings for filtered out ones + -- Return empty strings for filtered-out Pods return "", "" else - -- Process special Pods to make the returned workload type a supported type + -- For special Pods, convert the workload type to one we support workloadType = "StatefulSet" end - -- Get workloadName and ensure it is of string type + -- Get workloadName and ensure it is a string -- Here, the Pod has ownerReferences data, and the name in ownerReferences is the workloadName - -- If the Pod does not have ownerReferences data, you can calculate the workloadName based on the pod name + -- If the Pod has no ownerReferences data, you can derive workloadName from the pod name local workloadName = tostring(meteTable["name"] or "") return workloadType, workloadName end ``` -After uploading this plugin, you can extract the corresponding workload type and workload name for this Pod. +After uploading this plugin, you will be able to extract the corresponding workload type and workload name for this Pod. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/01-query/01-overview.md b/translate/translated/06-guide/01-ee-tenant/01-query/01-overview.md deleted file mode 100644 index 286186bc..00000000 --- a/translate/translated/06-guide/01-ee-tenant/01-query/01-overview.md +++ /dev/null @@ -1,23 +0,0 @@ ---- -title: Search -permalink: /guide/ee-tenant/query/overview/ ---- - -> This document was translated by ChatGPT - -# Search - -DeepFlow is a highly automated observability platform that provides application and network metrics, events, distributed tracing, and log data. With the AutoTagging capability, it injects unified attribute tags into all observability data. When dealing with these massive amounts of data, DeepFlow not only offers productized analysis capabilities but also features rapid search capabilities. As an efficient information retrieval tool, the search box can quickly and accurately achieve lookup and filtering when dealing with large volumes of data. This chapter will focus on how to use the DeepFlow search box. - -The DeepFlow search box can be divided into four main categories, and this chapter will provide a detailed analysis of each type of search box and its common application scenarios. - -- [Resource Search Box](./service-search/) -- [Path Search Box](./path-search/) -- [Log Search Box](./log-search/) -- [Metric Search Box](./metric-search/) - -In addition to defining the search box, DeepFlow also enhances rapid search capabilities and search-related configurations. - -- [Search Snapshots](./history/) -- [Left Quick Filter](./left-quick-filter/) -- [Search Box Configuration](../configuration/settings/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/01-query/02-service-search.md b/translate/translated/06-guide/01-ee-tenant/01-query/02-service-search.md deleted file mode 100644 index 0334430e..00000000 --- a/translate/translated/06-guide/01-ee-tenant/01-query/02-service-search.md +++ /dev/null @@ -1,115 +0,0 @@ ---- -title: Resource Search Box -permalink: /guide/ee-tenant/query/service-search/ ---- - -> This document was translated by ChatGPT - -# Resource Search Box - -The `Resource Search Box` is used in Application-Resource Analysis, Network-Resource Analysis, and Network-Resource Inventory. - -![01-资源搜索框](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664ac73e1e086.png) - -- **① Search Snapshot**: Refer to the [Search Snapshot](./history/) section for details. -- **② Search Input Form**: You can switch the form of search input. Currently, there are Free Search, Container Search, and Process Search. See the following sections for details. - -## Free Search - -![02-自由搜索](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405156644260e09259.png) - -- **① Search Condition Input Box**: Supports both Chinese and English associative input, and supports Tags in the data table as search conditions. -- **② Clear Search Conditions**: Clears the `Search Condition Input Box`. -- **③ Switch Main Group**: Resource grouping, corresponding to `Resource` in the functional interface. -- **④ Switch Subgroup**: Other groupings, corresponding to `Group Attributes` in the functional interface. - -In the `Search Condition Input Box`, each complete search condition is called a `Search Tag`. The following explains how to manage `Search Tags`. - -![03-搜索标签](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa57a56f.png) - -![04-操作符](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa702aed.png) - -![05-候选项](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c50ecc63c1.png) - -- **① Tag Name**: Supports querying Tags in the data table. For detailed descriptions, see `Database Fields`. - - Supports both Chinese and English associative input. - - Hover over the Tag Name to view detailed information. - - Semantics: Different Tags are connected using `and`, while the same Tag uses different logical operators based on the `operator`. - - a: `=` , `:` , `~` are connected using `or`. - - b: `!=` , `!:` , `!~` are connected using `and`. - - c: `>=` , `<=` , `>` , `<` are connected using `and`. - - After connecting a/b/c, they are further connected using `and`. For example: `Search Condition Input Box: server_port > 20, server_port < 80, server_port != 44, server_port != 45`, the effective condition is `(server_port > 20 and server_port < 80) and (server_port != 44 and server_port != 45)`. -- **② Operator**: Currently supports exact match, fuzzy match, and regex match. - - Exact Match: Corresponds to `=` , `!=` , `>=` , `<=` operators. For `resource type` Tags, it matches by resource ID; for others, it matches by actual input. - - Fuzzy Match: Corresponds to `:` , `!:` operators. String matching supports `*` wildcard. For example, `*123*` matches all strings that contain `123`, while `123` matches strings that exactly equal `123`. - - Regex Match: Corresponds to `~` , `!~` operators. String regex matching. -- **③ Tag Value**: Filter or directly input the value to be filtered. - - NULL: Null value, generally used with `!=` to filter `all`. - - **⑦ Table Filter**: When there are duplicate names in the candidate options or multiple selections are needed, use `Table Filter` for precise resources. -- **④ Disable**: Disable the search condition corresponding to the current `Search Tag`. -- **⑤ Modify**: Modify the search condition corresponding to the current `Search Tag`. -- **⑥ Delete**: Delete the current `Search Tag`. - -## Container Search - -The container search form fixes commonly used resource Tags in container scenarios as dropdown forms for quick filtering of container resources. - -![06-容器搜索](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240515664425a2b3c16.png) - -- **① Container Resource Dropdown**: Click the dropdown to quickly select the container resources to be filtered. The options in the dropdown can be linked with the previous selections. -- **② Search Condition Input Box**: See the description in the `Free Search` section above. -- **③ Collapse Search Condition Input Box**: Click to quickly collapse the `Search Condition Input Box`. -- **④ Switch Group**: Quickly switch container resource Tags. - -## Process Search - -The process search form is similar to the container search, mainly fixing commonly used process-related Tags as dropdown forms for quick filtering of process resources. - -![07-进程搜索](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240515664426411eea4.png) - -# Application Scenarios - -## View Service Performance of a Workload - -- Functional Page: Application-Metrics -- Search Tag: pod_ns = gcp-microservices-demo -- Main Group: auto_service -- Subgroup: -- - -![05-查询结果](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa039078.png) - -## View Performance of a Specific Workload in a Namespace - -- Functional Page: Application-Metrics -- Search Tag: pod_ns = gcp-microservices-demo, pod_group : loadgenerator -- Main Group: auto_service -- Subgroup: -- - -![06-查询结果](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa17b7c6.png) - -## View Top 5 Cloud Servers by Traffic - -- Functional Page: Network-Service -- Search Tag: None -- Main Group: chost -- Subgroup: -- - -![07-查询结果](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa2642e9.png) - -## View Top 5 Server Ports by Traffic on a Cloud Server - -- Functional Page: Network-Service -- Search Tag: role = server -- Main Group: chost -- Subgroup: server_port - -![08-查询结果](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa2adfda.png) - -## View Network Performance of a Specific Port on a Cloud Server - -- Functional Page: Network-Service -- Search Tag: role = server, chost = cn-chengdu.172.16.0.196, server_port = 22 -- Main Group: chost -- Subgroup: server_port - -![09-查询结果](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa44b491.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/01-query/04-log-search.md b/translate/translated/06-guide/01-ee-tenant/01-query/04-log-search.md deleted file mode 100644 index 962c3ebd..00000000 --- a/translate/translated/06-guide/01-ee-tenant/01-query/04-log-search.md +++ /dev/null @@ -1,79 +0,0 @@ ---- -title: Log Search Box -permalink: /guide/ee-tenant/query/log-search/ ---- - -> This document was translated by ChatGPT - -# Log Search Box - -The `Log Search Box` is used in Application - Call Logs/Distributed Tracing and Network - Flow Logs. - -Compared to the `Path Search Box`, the `Log Search Box` only lacks the `grouping` capability. For detailed operation instructions, please refer to [Path Search Box](./path-search/). - -![00-Log Search Box](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f5e7a6cd.png) - -# Application Scenarios - -## View Abnormal Calls of a Service - -- Feature Page: Application - Call Logs - ---- - -- Service Set: S1 -- Search Tags: pod_service = frontend-external, response_status != normal -- Path: Intra-service, Inter-service, WAN -- Direction: Bidirectional - -![01-Query Results](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f61ad6e0.png) - -## View MySQL Calls of a Service - -- Feature Page: Application - Call Logs - ---- - -- Service Set: S1 -- Search Tags: pod_service = cars, l7_protocol = MySQL -- Path: Intra-service, Inter-service, WAN -- Direction: Bidirectional - -![02-Query Results](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f60c6540.png) - -## View Flow Logs of a Specific Five-tuple - -- Feature Page: Network - Flow Logs - ---- - -- Service Set: S1 -- Search Tags: pod = insurances-v1-d895774d6-26wf7, client_port = 46168, protocol = TCP -- Path: Intra-service -- Primary Group: pod -- Secondary Group: observation_point -- Direction: Client - ---- - -- Service Set: S2 -- Search Tags: pod = mysqldb-v1-5cc78df8d-fwrn4, server_port = 3306 -- Path: Intra-service -- Primary Group: pod -- Secondary Group: observation_point -- Direction: Server - -![03-Query Results](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f601adf9.png) - -## View Flow Logs of a POD with Connection Establishment Anomalies - -- Feature Page: Network - Flow Logs - ---- - -- Service Set: S1 -- Search Tags: pod = frontend-97cc49c74-qs6wh, close_type = connection-client ACK missing, close_type = connection-server SYN missing, close_type = connection-client port reuse, close_type = connection-server direct reset, close_type = connection-server other reset -- Path: Intra-service, Inter-service, WAN -- Direction: Bidirectional - -![04-Query Results](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b087b6e86.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/01-query/05-metric-search.md b/translate/translated/06-guide/01-ee-tenant/01-query/05-metric-search.md deleted file mode 100644 index 320d9956..00000000 --- a/translate/translated/06-guide/01-ee-tenant/01-query/05-metric-search.md +++ /dev/null @@ -1,23 +0,0 @@ ---- -title: Metric Search Box -permalink: /guide/ee-tenant/query/metric-search/ ---- - -> This document was translated by ChatGPT - -# Metric Search Box - -Currently, both the `Metrics Page` and `Chart-Edit-Search Conditions` use the `Metric Search Box`. - -![Metric Search Box](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f741fb51.png) - -- **① Database**: The database where the metric is located, such as Application, Network, Event, or Prometheus. -- **② Data Table**: The data table where the metric is located, such as the `Metrics (Minute Level)` or `Metrics (Second Level)` data table under the `Application` database. -- **③-⑦**: Please refer to the chapters [Service Search Box](./service-search/), [Path Search Box](./path-search/), and [Log Search Box](./log-search/). -- **⑧ Switch to PromQL Input Box**: Click the button to switch between `Simplified Search` and `PromQL Search` modes. -- **⑨ Metric Dropdown**: Select the metric you want to view. Note: You must select a metric. -- **⑩ Operator Dropdown**: Select the aggregation operator. For detailed explanations of operators, please refer to the document [Calculation Logic of Metric Operators](../../../features/universal-map/metrics-and-operators/#%E8%81%9A%E5%90%88%E7%AE%97%E5%AD%90). -- **⑪ Secondary Operator**: Select the secondary operator. -- **⑫ Disable/Enable**: Disable the metric to stop querying; enable the metric to initiate a query. -- **⑬ Add Metric**: Supports adding multiple metrics, each corresponding to a line chart. -- **⑭ Add Query**: Supports adding multiple query conditions. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/01-query/06-history.md b/translate/translated/06-guide/01-ee-tenant/01-query/06-history.md deleted file mode 100644 index f73de781..00000000 --- a/translate/translated/06-guide/01-ee-tenant/01-query/06-history.md +++ /dev/null @@ -1,51 +0,0 @@ ---- -title: Search Snapshots -permalink: /guide/ee-tenant/query/history/ ---- - -> This document was translated by ChatGPT - -# Search Snapshots - -The search snapshot feature records the user's past search-related information. It helps you record the current page's query conditions, query time, and chart configuration settings. It also supports quickly selecting search snapshots from a dropdown menu to apply to the page, sharing search snapshots, setting default load pages, and other functions. - -Next, we will introduce how to use the search snapshot feature. - -## Basic Introduction - -![00-Basic Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d6263c60c9.png) - -- **① Search Snapshot Dropdown:** Displays all saved search conditions for the current page in a dropdown menu and supports managing search snapshots. For detailed usage, please refer to [Search Snapshot Dropdown]. -- **② Save Search Conditions:** Click to save the current page's search conditions, time, and other information. For detailed usage, please refer to [Save Search Conditions]. - -## Search Snapshot Dropdown - -![01-Dropdown](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d626419381.png) - -The search snapshot bar consists of a search bar, dropdown menu, and description box. - -- Search Bar: Allows querying the name of the search snapshot, supporting both Chinese and English suggestions. - - Click the search bar to pop up the dropdown menu, displaying the saved search snapshots for the current page. -- Dropdown Menu: Displays the search condition records saved by the user on the current page and the search conditions shared by other users. - - Also supports starring, modifying, and other operations on search snapshots. For detailed usage, please refer to the [Manage Search Snapshots] section. -- Description Box: When the mouse hovers over a search snapshot, the description box displays related information. - - Shows the search snapshot's name, description, search count, permissions, source account, creation time, etc. - -## Save Search Conditions - -![02-Save Search Conditions](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d6264dba0a.png) - -Users can save the search conditions on the current page. Click the `Save` icon to edit the name and description of the save. It also supports remembering the search time range and the configuration of the Panel. - -## Manage Search Snapshots - -![03-Manage Search Snapshots](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d6265ca207.png) - -- **① Star:** Mark the search snapshot as important, displaying it first in the dropdown menu and table to help users find the search snapshot more quickly. - - Click again to unstar. -- **② Edit:** Modify the `name` or `description` of the search snapshot. -- **③ Share:** Share the search snapshot with one or more specified users, with options to assign `read-only` or `read-write` permissions, and display the share count. -- **④ Query:** Open the search snapshot conditions in a new page and display the query count of the search snapshot. -- **⑤ Set Default Load Page:** Click the icon to set the search snapshot conditions as the default load page for the current page. - - Once set successfully, the icon will be highlighted, and the setting will take effect when re-entering from other pages. -- **⑥ Delete:** Delete the search snapshot. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/01-query/07-left-quick-filter.md b/translate/translated/06-guide/01-ee-tenant/01-query/07-left-quick-filter.md deleted file mode 100644 index 75053cbd..00000000 --- a/translate/translated/06-guide/01-ee-tenant/01-query/07-left-quick-filter.md +++ /dev/null @@ -1,32 +0,0 @@ ---- -title: Left Quick Filter -permalink: /guide/ee-tenant/query/left-quick-filter/ ---- - -> This document was translated by ChatGPT - -# Left Quick Filter - -The left quick filter feature supports quick filtering of tag and metric fields. Let's take the Path Overview page as an example to demonstrate how to use the left filter. - -![00-Left Quick Filter](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a9fb1183e5.png) - -The left quick filter supports field filtering queries for the data in the table on the page, making it convenient for users to quickly search the data and improve query efficiency. Currently, the left quick filter only supports filtering for some fields, and other fields will be gradually opened in the future. - -## Usage Introduction - -Click the `Quick Filter Button` at the top left, and the panel on the left side of the page will expand to display the fields that can be filtered. When the mouse hovers over the data, you can view the explanation of the data and option values. When the left quick filter is effective, the query conditions in the page service search bar are synchronized. - -![01-Usage Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a9fb139c2f.png) - -- The page opens the left quick filter panel by default -- **Operation Instructions:** - - When the mouse hovers over the data item and the checkbox, it will prompt the state after clicking at the current position - - State Explanation: - - Select All: By default, all options under all fields are selected initially, and the page query does not perform any filtering - - Select Only This Item: Clicking the row of the option indicates that only the current option value of the field is selected as the query condition - - Deselect: Clicking the row of the option indicates canceling the `Select Only This Item` state, reverting to `Select All` - - Toggle State: Clicking the checkbox indicates `selecting` or `deselecting` the option - - Select: The checkbox is checked, indicating that the option is included in the query - - Deselect: The checkbox is unchecked, indicating that the option is not included in the query - - Clear Filter: Click the clear filter icon at the top right of the data panel to clear the value filter for that field, reverting to `Select All` \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/02-list.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/02-list.md deleted file mode 100644 index 60a8e208..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/02-list.md +++ /dev/null @@ -1,26 +0,0 @@ ---- -title: Dashboard List -permalink: /guide/ee-tenant/dashboard/list/ ---- - -> This document was translated by ChatGPT - -# Dashboard List - -The dashboard list page displays all the dashboards created by the current user and some basic operations. - -![Overview.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240514664305b709a65.png) - -- Dashboards are divided into two categories: Custom Dashboards and Built-in Dashboards - - Custom Dashboards: Displays dashboards created by the current account and those editable within the team organization - - Built-in Dashboards: Visual dashboards provided by DeepFlow for system information, not editable or deletable -- **① New Dashboard:** Click to create a new dashboard, enter the name of the new dashboard, and you can create your dashboard. You can add a description as needed -- **② Import Dashboard:** Click to import a dashboard, you can define a name and select a JSON file to import. Note: Currently, only JSON dashboard files exported by DeepFlow are supported. For details, please refer to `Export` -- **③ Search:** Supports entering any string in the search bar, such as name, team, description, creator, latest modification time, etc., to match the list information -- **④ Settings:** You can set the display method of column width, such as evenly distributed column width or content-based column width -- **⑤ Delete:** Supports batch deletion of all selected dashboards -- **⑥ Export:** Supports batch export of all selected dashboards -- **⑦ Star:** Starred dashboards will be displayed with priority -- **⑧ More:** Includes edit, export, and delete functions - - Edit: Supports modifying the name and description of the dashboard - - Export: Supports exporting the dashboard in JSON format \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/03-use.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/03-use.md deleted file mode 100644 index ee84b4b6..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/03-use.md +++ /dev/null @@ -1,72 +0,0 @@ ---- -title: Dashboard Details -permalink: /guide/ee-tenant/dashboard/use/ ---- - -> This document was translated by ChatGPT - -# Dashboard Details - -The dashboard details page displays user-customized visualization panels. - -![00-Dashboard Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031165eec281b77db.png) - -- **① Dashboard Dropdown:** The dropdown options are the names of the dashboards. Selecting a name allows for quick switching between dashboards. -- **② Query Area:** Supports quick one-click switching of the chart query area, and you can also switch the query area on the chart. -- **③ Time Picker:** Allows customization of the time range for the visualization panel data. For more details, please refer to the [Time Picker] section. -- **④ Time Interval:** Allows selection of the time granularity for data aggregation. Note: Aggregation time granularity is only effective for time series charts: TOP N line charts/line charts/trend analysis charts. - - Second level: 1, 5, 10, 30s - - Minute level: 1, 5, 10, 30m - - Hour level: 1, 3, 6, 12h - - Day level: 1, 7d -- **⑤ Manual Refresh:** Click the refresh button to refresh the data in real-time. -- **⑥ Auto Refresh:** Auto-refresh is disabled by default. You can choose to enable auto-refresh at 1m or 5m intervals. -- **⑦ Add Chart:** Supports adding charts and groups. For more details, please refer to the [Add Chart] section. -- **⑧ Full Screen:** Click the button to display the current dashboard in full screen. Press `Esc` to exit full screen. -- **⑨ Save:** After modifying the time range, time interval, topology position, chart configuration, template variable values, query area, etc., click the save button to save the changes. If you want to save it as a copy, you can check the save as option. -- **⑩ Settings:** In the settings option, you can export, delete, manage template variables, and perform a series of operations on the dashboard. For more details, please refer to the [Settings] section. - -## Time Picker - -The time picker supports users in viewing historical data of the dashboard using either `absolute time` range or `relative time` range. - -![01-Time Picker](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031165eec28050664.png) - -- Absolute Time: - - Supports selecting a time range from the calendar. - - Supports manually entering a time range in the `YYYY-MM-DD HH-mm-ss` format. -- Relative Time: - - Supports shortcuts for quickly selecting relative time. - - Last 5m, 15m, 30m, 6h, 1d, 7d, 30d, etc. -- Supports entering relative time using the `now` keyword. - - `now`: Corresponds to the exact current time in year, month, day, hour, minute, and second. - - `now/d`: Represents today. If used as the start time, it corresponds to 00:00:00 of today; if used as the end time, it corresponds to 23:59:59 of today. - - `now-$num d`: Represents the last num days. The exact time corresponds to the current time minus the specified number of days, where num is an integer between 1 and 100. -- Search Snapshot Records: Records historical search times for quick viewing. - -## Add Chart - -There are two ways to add charts to the dashboard: you can add charts within the dashboard or add charts from external pages to the dashboard. - -![02-Add Chart in Dashboard](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240514664327cf0b5d6.png) - -- Click the `Add Chart` button to add various types of charts such as line charts, bar charts, pie charts, overview charts, traffic topology, distribution charts, tables, text, etc. Grouping is also supported. -- For adding charts from external pages, please refer to the [Add Chart](./add-panel/) section. - -## Settings - -The settings button on the dashboard page is equipped with a variety of operational functions to facilitate better use of the dashboard. - -![03-Settings](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031165eec3a58224c.png) - -- Set Global Data Table: Quickly switch the data table referenced by the charts within the dashboard. -- Manage Template Variables: Quickly change the search conditions of the charts. For more details, please refer to the [Template Variables](./variable-template/) section. -- Enable/Disable Tip Sync: When enabled, you can simultaneously view data information at the same time point for all time series-related charts. - - Time series-related charts include: line charts, trend analysis charts. -- Create Module: Allows charts to be categorized and placed by module. Modules can be collapsed and expanded as needed, and names can be modified or modules deleted in the module bar. -- Switch Fill Mode: When data does not exist at a certain time point, you can switch the fill mode to handle it as needed. - - Fill 0: Fill the current time point with 0. - - Fill null: The current time point data is empty. - - Fill none: Exclude the current time point. -- Switch Tile/Stack: Quickly switch the display form of all time series-related charts on the dashboard details page between tile and stack. -- Full/Default Name Display: Quickly switch the name display mode of all charts on the dashboard details page. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/04-add-panel.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/04-add-panel.md deleted file mode 100644 index 1d1744fb..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/04-add-panel.md +++ /dev/null @@ -1,24 +0,0 @@ ---- -title: Add Panel -permalink: /guide/ee-tenant/dashboard/add-panel/ ---- - -> This document was translated by ChatGPT - -# Add Panel - -A Dashboard is composed of one or more `panels`. This chapter will introduce how to add `panels` to a `dashboard`. - -Currently, it only supports adding `panels` to the `dashboard` from the feature pages, including charts from the `Metrics`, `Applications`, and `Network` feature pages. - -**Step 1**: Navigate to the feature module page where you want to add a panel, for example, click `① Applications - Services` to enter the `Service Overview` page. - -![Step 1](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650824ebccb7c.png) - -**Step 2**: Select the panel you want to add to the dashboard, click the `② Settings` button on the panel, and choose the `③ Add to Dashboard` option. - -![Step 2](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650824ed30950.png) - -**Step 3**: Confirm the `④ Name` of the panel and select the `⑤ Dashboard` to which the panel should be added. Click `⑥ Confirm` to successfully add the panel to the corresponding dashboard. - -![Step 3](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650824edae26e.png) diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/05-variable-template.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/05-variable-template.md deleted file mode 100644 index f926b9a7..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/05-variable-template.md +++ /dev/null @@ -1,110 +0,0 @@ ---- -title: Template Variables -permalink: /guide/ee-tenant/dashboard/variable-template/ ---- - -> This document was translated by ChatGPT - -# Template Variables - -Template variables allow you to quickly change the search criteria of charts by defining a set of variables in the current dashboard and referencing these variables in the chart's search conditions. This way, you can view the corresponding dashboard by only changing the variable values, without needing to create multiple identical visualization panels just because of different search conditions. - -## Managing Template Variables - -Template variables can be managed uniformly through the `Template Variable List`. - -As shown in the figure below, the template variable list supports operations such as `① Add`, `② Delete`, and `③ Modify`. You can also enter any string in the `⑤ Search Bar` and adjust the display mode of the `④ Column Width`, such as evenly distributing column width or allocating column width based on content. - -![00-Template Variable List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf59d3ef4d.png) - -## Creating Template Variables - -When you need to build some quick search conditions for the current dashboard, you can `Create Template Variables` to achieve this. For example, when building an `Application Observability Dashboard`, you may need to quickly view the dashboards of different `applications`, so you can create a `template variable` for the `application` search condition. - -**Step 1**: Click the `① Settings` button on the dashboard details page and select `② Manage Template Variables`. - -![01-Step 1](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf68c0b038.png) - -**Step 2**: Click the `③ Create Template Variable` button in the template variable list popup. - -![02-Step 2](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf70be224b.png) - -**Step 3**: Create the template variable as needed. The DeepFlow platform provides three types of template variables: `Dropdown`, `Text Input`, and `Group`. Detailed descriptions can be found in the corresponding sections below. - -![03-Step 3](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf7bb39bbe.png) - -### Dropdown - -Dropdown type template variables change the search conditions through a dropdown menu. Currently, this type of template variable supports the `resource` and `xx_enum` types of `Tags` in the DeepFlow platform database. - -- Note: For a description of the DeepFlow platform database, see the subsequent sections. - -![04-Dropdown](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf81b3b73b.png) - -- **① Query Type:** Different data transmission methods - - Query by ID: Supports ID-based queries - - Query by Name: Supports string-based queries -- **② Value Range:** Supports setting the value range of template variables in two ways: `Static Values` and `Dynamic Values` - - Static Values: The value range is fixed after being referenced - - Dynamic Values: Compared to static values, the value range of dynamic values can be influenced by `Static Values` or `Value Tag`. For usage details, refer to the [Creation and Reference] section - - **③ Data Source:** Determines the data table where the template variable values are located - - **④ Value Tag:** Determines the Tag corresponding to the template variable values - - **⑤ Value Range:** Selects the values corresponding to the template variable -- **⑥ Selection Mode:** Changes the search conditions through a dropdown menu, default is single selection - - Multi-select: Check `Multi-select` to switch to multi-select mode - - Select All: Check `Select All` to include an `All` option in the candidates, selecting all values of the current template variable -- How to Reference: When adding query conditions in the chart's search bar, enter the Tag, and the established template variables with the same `Tag` will appear as candidates in the dropdown menu. For usage details, refer to the [Creation and Reference] section - -#### Creation and Reference - -Next, we will demonstrate how to create and reference `Static Template Variables` and `Dynamic Template Variables`, and how to link dynamic and static template variables. - -- First, create a static template variable named `K8s Namespace` for `pod_ns`, with a value range of `deepflow-ebpf-istio-demo, deepflow-otel-grpc-demo, deepflow-telegraf-demo`. - -![05-Create Static Template Variable](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd4b0c94b.png) - -- Next, create a dynamic template variable named `K8s Workload` for `pod_group`, with a value range of `pod_ns = K8s Namespace`, meaning the dropdown candidates for `pod_group` will change based on the selection of `pod_ns`. - -![06-Create Dynamic Template Variable](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd4c9cee2.png) - -- Then, select the query condition `pod_group = K8s Workload` and reference the template variable in the chart's search conditions. - -![07-Template Variable Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd4e0596a.png) - -- Finally, you can quickly switch the referenced template variables at the top of the dashboard. Different selections for `K8s Namespace` will result in different candidates for `K8s Workload`. - -![08-Use Template Variable](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd504e2f4.png) - -![09-Switch Template Variable](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd5108331.png) - -### Text Input - -Text input type template variables change the search conditions by entering a string. - -![10-Text Input](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091865082716002b8.png) - -Text input type template variables can be referenced by any `Tag` or `Operator` that can be directly entered. The form in which they appear in the search conditions is similar to that of `Dropdown` type template variables. - -- ① Tag Reference: Supports int, int_enum, string, ip, mac types. Tags of these data types support all operators. - -![11-Tag Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309186508271402080.png) - -- ② Operator Reference: Supports :, !:, =~, !~ types. These types support all Tag types. - -![12-Operator Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091865082716add4e.png) - -### Group - -Group type template variables change the grouping through a dropdown menu. For example, when you need to drill down data layer by layer from `K8s Cluster` -> `K8s Namespace` -> `K8s Container Service` -> `K8s Workload` -> `K8s Container POD`, you can create this type of template variable. - -![13-Group](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650827184c5b7.png) - -- Value Range: Currently, all `Tags` in the DeepFlow platform database can be used as values for this type of template variable. - - ①: Determines the data table where the template variable values are located - - ②: Selects the values corresponding to the template variable -- Selection Mode: For detailed description, see the Dropdown type template variable description - - Note: The main group cannot reference template variables in `Multi-select` or `Select All` mode - -Group type template variables can only be referenced by groupings in search conditions. They appear as candidates in the grouping dropdown menu. - -![14-Group Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091865082715de5ff.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/06-right-slide-box.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/06-right-slide-box.md deleted file mode 100644 index 06d86e7f..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/06-right-slide-box.md +++ /dev/null @@ -1,40 +0,0 @@ ---- -title: Edit Right Slide Box -permalink: /guide/ee-tenant/dashboard/right-slide-box/ ---- - -> This document was translated by ChatGPT - -# Edit Right Slide Box - -The right slide box of the dashboard can be edited as needed to add or remove `system panels` and `dashboard panels`. The edited panels are saved and remembered for each chart in the dashboard. - -- System Panels: Predefined panels by the DeepFlow system. For details, refer to [Tracing - Right Slide Box](/guide/ee-tenant/tracing/right-sliding-box/) -- Dashboard Panels: Custom panels added by users in the `dashboard` - -![Right Slide Box Panels](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664579a8512bb.png) - -- **① Right Slide Box Panels:** The display position of the panels, showing all `system panels` by default -- **② Panel Management:** Allows `sorting`, `deleting`, and `editing` operations for dashboard panels -- **③ Add Panel:** See the following sections for details - -## Add Panel - -You can add `system panels` and `dashboard panels` - -### Add System Panel - -![Add System Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664579b508cbe.png) - -**① Select Panel:** Quickly add or remove a system panel by clicking the checkbox. - -### Add Dashboard Panel - -![Add Dashboard Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664579aab5031.png) - -- **① Dashboard Name:** Select the name of the dashboard to be added to the right slide box panel. Duplicate additions are not allowed. Once added, the dashboard panel is loaded in the right slide box in `read-only` mode. -- **Associated Conditions:** Set the values read by the template variables when the dashboard is loaded in the right slide box - - **② Template Variables:** Template variables in the dashboard - - **③ Template Variable Values:** Set the values for the template variables, which can be: - - Default Variable Value: When the dashboard loads, it reads the default value set for the template variable in the dashboard - - $Tag: When the dashboard loads in the right slide box, it reads the value of the Tag corresponding to the search condition. For example, if the search condition carries `protocol = tcp` when entering the right slide box, the `protocol` template variable will be set to `tcp` when the page loads. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/01-overview.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/01-overview.md deleted file mode 100644 index 0b65840d..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/01-overview.md +++ /dev/null @@ -1,22 +0,0 @@ ---- -title: Overview -permalink: /guide/ee-tenant/dashboard/panel/overview/ ---- - -> This document was translated by ChatGPT - -# Overview - -Charts are the smallest unit of visualization components in DeepFlow. They serve as visualization components for each functional page and can also be added to a Dashboard to create custom visual views. Each chart can be operated independently, and users can modify the search conditions, metrics, styles, etc., according to their needs. - -DeepFlow supports the use of various charts. Next, we will introduce how to use the following charts. - -- [Traffic Topology](./topology/) -- [Distributed Tracing Flame Graph](./flame/) -- [Line Chart](./line/) -- [Bar Chart](./bar/) -- [Pie Chart](./pie/) -- [Histogram](./histogram/) -- [Table](./table/) -- [Overview Chart](./stat/) -- [Text](./text/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/02-topology.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/02-topology.md deleted file mode 100644 index 211517d2..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/02-topology.md +++ /dev/null @@ -1,153 +0,0 @@ ---- -title: Traffic Topology -permalink: /guide/ee-tenant/dashboard/panel/topology/ ---- - -> This document was translated by ChatGPT - -# Traffic Topology - -DeepFlow's traffic topology is used to display the dependencies between services or resources, facilitating better analysis and problem-solving, such as analyzing performance bottlenecks, single points of failure, or potential dependency access issues. - -## Overview - -![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031465f2d3476b5c0.png) - -The traffic topology consists of `nodes`, `paths`, and some operations: - -- **① Node:** Represents a service or resource, corresponding to the `group` in the search criteria. It can be a container service, cloud server, or region, etc. -- **② Path:** Represents the direction of service or resource, where the `client` accesses the `server`. -- **Operations:** You can hover over or click on `nodes` or `paths`. - - Hover: Highlight the `node` or `path` to view metrics. - - Click: View details of the `node` or `path` in a right-slide frame. - -### Topology Details - -![01-Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031465f2cfe0cb741.png) - -- **① Switch Query Area:** If there are multiple storage areas in the previous data, you can quickly switch areas for data queries. - - Note: If the query criteria are not grouped, there is no `Switch TOP Data` function. -- **② Switch Top Data:** Sort the main metric values of the nodes in descending order based on the grouping. - - Note: If the query criteria are not grouped, there is no `Switch TOP Data` function. -- **③ Expand Table:** Click to expand or close the table. For details, please refer to the [Expand Table] section. -- **④ Modify Metrics:** You can modify the metrics. For details, please refer to the [Modify Metrics] section. -- **⑤ Settings:** Click to set the `traffic topology`. For details, please refer to the [Settings] section. -- **⑥ Delete:** This is a `Dashboard` capability. If you do not need to display this `traffic topology` in the Dashboard, you can click the delete button to remove it. -- **⑦ Manually Supplement Resource Relationship Mode:** In this mode, you can manually add paths between nodes. -- **⑧ Waterfall/Free Topology:** Supports switching the display form of the topology. Free topology is generally used for scenarios with many nodes and complex paths; waterfall topology is generally used for scenarios with fewer nodes and simpler paths. -- **⑨ Auto Layout:** The system arranges the nodes in a tree structure based on the path access relationships. -- **⑩ Random Layout:** The system arranges the nodes in a star structure. -- **⑪ Save Topology:** This is a `Dashboard` capability. After modifying the `traffic topology`, you can choose the `Save Topology` button to save the changes, such as remembering the time range, topology position, topology configuration, and variable template values. -- **⑫ Legend:** You can open the legend to view the meanings of icons and lines. For details, please refer to the [Legend] section. - -### Hover TIP - -![02-Hover TIP](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7b8291e1be.png) - -When the mouse hovers over a `node`, the `path` associated with the `node` is automatically highlighted. When hovering over a `path`, the `nodes` associated with the `path` are automatically highlighted. At the same time, you can view the metrics in the form of a TIP, which can display the metrics corresponding to different `observation points`. - -### Hover Display Metrics - -![03-Hover Display](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7b826aff77.png) - -Taking the mouse hovering over a `path` as an example, the TIP display content is introduced. - -- The first row: Explanation and legend display area. The legend explanation is as follows: - - Application Function - - A Application: Represents metrics obtained through application instrumentation, currently indicating data with `signal source=OTel`. - - S System: Represents metrics obtained through eBPF, currently indicating data with `signal source=eBPF`. - - E Endpoint Network: Represents metrics obtained through traffic capture (BPF), currently indicating data collected from the network card of the client or server. - - M Middle Network: Represents metrics obtained through traffic capture (BPF), currently indicating data collected from locations other than the network card of the client or server. - - Network Function: All metrics are obtained from traffic capture (BPF). - - D Network Card: Indicates data collected from the network card of the client or server. - - K Container Node: Indicates data collected from the network card of the container node. - - H Host: Indicates data collected from the network card of the host. - - M Middle Network: Indicates data collected from locations other than the above network cards. - - Corner Mark: Distinguishes whether the current data collection location is on the client or server. - - C: Represents the client. - - S: Represents the server. -- The second row: Hover `node/path` name information. -- Others: Metrics display area. - - Displays the relevant metric values based on the data location. - - The last column shows the difference of all `observation points`, which can quickly determine if there is data inconsistency in `sent traffic`. - - Example: As shown in the figure below, it indicates the endpoint network data of the client. - ![04-Icon](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509427e1c1c9.png) - -### Settings - -Users can click the gear icon to set the `traffic topology`. - -![05-Settings](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7b8295063a.png) - -- **Display Full Name:** Display the full name of the node or the default display. -- **Download CSV Data:** Supports downloading the data information of the `traffic topology`. -- **View API:** View the interface information that generates the `traffic topology`. - -### Expand Table - -Click the `Expand Table` button to display the metrics of `nodes` and `paths` in the `traffic topology` in a list form, including resource monitoring, path monitoring, and path difference tables. - -![06-Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965091707f4009.png) - -- Resource Monitoring: Displays the metrics of all nodes in the `traffic topology`. -- Path Monitoring: Displays the metrics of all paths in the `traffic topology`. -- Path Difference: Displays the difference in metrics of all paths in the `traffic topology` at different `observation points`. -- Search: Supports quick search and lookup of table data. -- Settings: Set the display method of column width, such as evenly distributing column width or allocating column width according to content. - -### Modify Metrics - -Metrics are one of the important components of the chart. DeepFlow provides a shortcut to help users quickly select metrics. You can choose the metrics to be displayed in the dropdown menu. - -![07-Modify Metrics](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509170932512.png) - -- **① Metric Name:** Select the metric name to display the metric data in the chart. -- **② Set as Main Metric:** Click the icon to set the metric as the main metric. When `Switching TOP Data`, the chart is sorted by the main metric value. -- **③ Advanced Settings:** You can add, delete, or modify metrics. For details, please refer to the [Advanced Settings] section. -- **Multi-select/Single-select:** Supported by some charts. In `multi-select`, the TIP in the chart can display multiple metrics. In `single-select`, only one metric can be displayed. - -### Advanced Settings - -For further settings of metrics, you can click `Modify Metrics -> Advanced Settings` to enter the settings page. As shown in the figure below, it supports users to add or delete metrics, add aggregation functions, modify display names, set thresholds, select metric templates, and other operations. - -![08-Advanced Settings](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965091709aebea.png) - -- **① Select Template:** Select `metric template` to quickly switch the metric template in the current popup window. -- **② One-click Clear:** One-click clear all metrics in the current popup window. -- **③ Save Template:** Save the settings state of the metrics in the current popup window as a `metric template`. -- **④ Metric Column:** Click the input box to pop up the dropdown box, and you can select metrics according to the classification. - - **Aggregation - Primary Operator:** Perform function aggregation operations on metrics, supporting functions such as average, sum, maximum, minimum, etc. - - **Aggregation - Secondary Operator:** Perform secondary calculations based on the data obtained by the `primary operator`. - - **Name:** Set the display name of the metric. - - **Unit:** Set the display unit of the metric. - - **Threshold:** Set the threshold of the metric. When the threshold is exceeded, the `node` or `path` will turn red to prompt. -- **⑤ Enable/Disable Metrics:** Show/hide the corresponding metrics. -- **⑥ Add Metrics:** Add `④ Metric Column` in the current popup window. - -### Edit - -The topology edit box consists of three parts: `① Chart`, `② Search Criteria`, and `③ Style and Settings`. - -![09-Edit](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031365f175fb51d9f.png) - -- **① Chart:** The chart is drawn based on `② Search Criteria` and `③ Style and Settings`. -- **② Search Criteria:** For the use of search criteria, please refer to the [Search](../../query/overview/) section. -- **③ Style and Settings:** Set the style, color, and other settings of the chart. - - **Style:** Rich functions support setting the style of the chart. - - **Title:** Supports modifying the chart name. - - **Chart Style:** - - Full Name Display: Full name display/abbreviated display of node names. - - Display Main Metric: When enabled, the main metric value will be displayed on the path. - - Close Thumbnail: Supports enabling/disabling the thumbnail. - - Turning Point of Step Line Chart: Supports setting the position of the data turning point. - - Area Fill: Supports filling the area between the line and the coordinate axis with color. - - Data Stacking: Supports displaying multiple data series in a stacked or tiled manner. - - Stacking: Displays the values of multiple data series in the same coordinate system in a stacked manner to show their overall trend and respective contributions. - - Tiling: Displays the values of multiple data series in the same coordinate system in a tiled manner to better show the differences and relationships between them. - - **Color:** Supports setting the color of paths and nodes. - -### Legend - -Click `Legend` to view the meanings of icons and lines. - -![10-Legend](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509170b4c72e.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/03-flame.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/03-flame.md deleted file mode 100644 index da2701b5..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/03-flame.md +++ /dev/null @@ -1,139 +0,0 @@ ---- -title: Distributed Tracing -permalink: /guide/ee-tenant/dashboard/panel/flame/ ---- - -> This document was translated by ChatGPT - -# Distributed Tracing - -DeepFlow presents application Spans, system Spans, and network Spans involved in a single call on a flame graph through `distributed tracing`, enabling collaboration across multiple departments such as business development teams, framework development teams, service mesh operations teams, container operations teams, DBA teams, and cloud operations teams on a single platform. - -## Overview - -Initiate a `tracing` operation on a call in the `distributed tracing` feature page, and then display it in the form of a right slide-out panel. This diagram shows the link call tracing, as shown below. - -``` -Note: The flame graph and topology graph of distributed tracing do not currently support adding to the Dashboard. -``` - -![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466431461b3f38.png) - -The call tracing right slide-out panel is divided into three parts: header information, data visualization, and call information data list. - -- **① Header Information:** Displays basic information of the link, such as client, server, request start time, duration, request type, request resource, etc. -- **① Data Visualization:** Displays call tracing Span data in the form of a flame graph or displays the services of call tracing in the form of a topology graph. -- **② Call Information Data List:** Displays associated information of the call. - -### Flame Graph - -![01-Flame Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965095885c540d.png) - -The flame graph consists of multiple `bar segments`, each representing a Span. The x-axis represents time, and the y-axis represents the depth of the call stack, displayed from top to bottom in the order of Span calls. Below is a detailed introduction: - -- **Length:** Combined with the x-axis, it represents the execution time of a Span, with both ends corresponding to the start and end times. -- **Service List:** Displays the proportion of time delay consumed by each service. Clicking on a service can link with the `flame graph` to highlight the corresponding Span of the service. - - **Color:** Application Spans and system Spans represent each service with a different color; all network Spans are gray (as network Spans do not belong to any service). -- **Display Information:** The `display information` of the bar segment consists of `icon` + `call information` + `execution time`. - - Icon: Different types of Spans are distinguished by icons. - - A: Application Span, collected through the Opentelemetry protocol, covering business code and framework code. - - S: System Span, collected through eBPF with zero intrusion, covering system calls, application functions (such as HTTPS), API Gateway, and service mesh Sidecar. - - N: Network Span, collected from network traffic through BPF, covering container network components such as iptables, ipvs, OvS, and LinuxBridge. - - Call Information: The `call information` displayed by different Spans varies slightly. - - Application Span and System Span: `Application Protocol`, `Request Type`, `Request Resource`. - - Network Span: `Observation Point`. - - Execution Time: The total time consumed from the start to the end of the Span. -- **Operation:** Supports `hover` and `click`. - - Hover: Hover over a Span to display `call information` + `instance information` + `execution time` in the form of a TIP. - - Instance Information: Application Span displays `service` + `resource instance`; System Span displays `process` + `resource instance`; Network Span displays `network card` + `resource instance`. - - Execution Time: Displays the entire execution time of the Span, i.e., the proportion of its own execution time. - - Click: Click on a Span to highlight itself and its parent Span, and view detailed information of the clicked Span. -- **Collapse Sidebar:** Click to collapse the `service list`. - -### Call Topology Graph - -![02-Call Topology Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965095886aa8de.png) - -The call topology graph displays data in an orderly and structured manner, with data aggregated by service as nodes. The parent-child relationships between Spans are displayed using horizontal and vertical lines, showing their request call relationships. Below is a detailed introduction: - -- **Node:** Corresponds to the service in the service list of the flame graph, aggregating one or more Spans under the same service into a node and displaying the time consumed by the service in the call chain. - - Display Information: The square node `display information` consists of `icon` + `call information` + `self time`. - - Icon: Different types of Spans are distinguished by icons. For details, please refer to the [Flame Graph] section. - - Self Time: The total time consumed by one or more Spans corresponding to the service. -- **Path:** Draws the topological path corresponding to the `parent Span` to `child Span` relationship in the flame graph. -- **Operation:** Supports `hover` and `click`. For details, please refer to the [Flame Graph] section. - -### Bottom Tab - -#### Call Details - -Displays detailed information of Spans in the flame graph in the form of a list. Clicking on a Span in the flame graph will highlight the corresponding call details in the list; conversely, clicking on a row in the list will highlight the corresponding Span. - -![Call Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643145809589.png) - -#### IO Events - -When clicking on a system Span in the flame graph, if the process corresponding to the system Span has IO read/write events, the corresponding IO events can be viewed. The IO events tab allows for quick viewing of the time consumed by Span for file read/write. - -![IO Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643145f4f784.png) - -**① First Row:** Overlays all IO event blocks of the threads below, with darker colors indicating more overlap. -**② Thread Row:** Displays the IO events of each thread, with each block corresponding to an event. The length of the block is calculated based on the start and end times of the IO event. - -- Tip: Consists of `file name` + `IO event type` + `event duration`. - **③ Detailed Information:** Displays details of the IO event. - -#### Flow Logs - -When clicking on a network Span in the flame graph, analyze the latency data of flow logs corresponding to the time period of the call log. - -![Flow Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643145d06086.png) - -**① Status Row:** Determines the observation point, flow duration, and flow log status. -**② Latency:** Analyzes network-related latency, including TCP connection latency, TLS connection latency, average data latency, average system latency, and average client wait latency. The calculation method of latency can be referred to in the `metric diagram`. - -#### Span Tracing - -When analyzing why a Span exists in the flame graph, the Span tracing feature can be used. Clicking on a Span in the flame graph displays the relationship with other Spans in the form of a list. DeepFlow's distributed tracing is calculated based on a series of IDs, including TraceID, SpanID, ParentSpanID, request X-Request-ID, response X-Request-ID, request Syscall TraceID, response Syscall TraceID, request TCP Seq number, and response TCP Seq number. When there is an association between IDs, the Spans can be displayed in a single flame graph, with the association of IDs marked in purple in the list. - -![Span Tracing](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466431459e1b6e.png) - -**① Clicked Span:** The Span clicked in the flame graph. -**① Associated Span:** The Span associated with the clicked Span. - -### Quick Understanding of Flame Graph - -![Flame Graph Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d2abc86b21.png) - -The flame graph represents the passage of time from left to right. In the sample call chain above, the complete processing of a business request goes through the following process: - -- (1) The "Client" process initiates an HTTP GET request, which is transmitted through multiple network cards to the "Frontend Service". -- (2) The "Frontend Service", to complete this business process, first initiates a DNS query to the "DNS Service", which is transmitted through the network to the "DNS Service". -- (3) The "DNS Service" processes the query and returns a DNS response to the "Frontend Service", which is transmitted through the network to the "Frontend Service". -- (4) The "Frontend Service" continues to initiate an SQL query, which is transmitted through the network to the "MySQL Service". -- (5) The "MySQL Service" processes the query and returns an SQL response to the "Frontend Service", which is transmitted through the network to the "Frontend Service". -- (6) The "Frontend Service" continues to initiate an RPC request, which is transmitted through the network to the "RPC Service". -- (7) The "RPC Service" processes the request and returns an RPC response, which is transmitted through the network to the "Frontend Service". -- (8) The "Frontend Service" receives the RPC response and replies with the final HTTP response to the "Client", which is transmitted through the network to the "Client". - -The difference in length between any two Spans represents the amount of delay introduced between the two positions. - -### Flame Graph Analysis Examples - -- **Example 1: Significant Difference Between Network Spans** - -In the figure below, the significant difference between two network Spans indicates a noticeable delay in the transmission of call data packets between two network cards. If the two network cards are "client container node" and "server container node", it indicates that the root cause of the slow response is the forwarding network between the container nodes. - -![Slow Call Flame Graph Example 1 - Significant Difference Between Network Spans](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d2926ceffc.png) - -- **Example 2: Significant Difference Between System Spans** - -In the figure below, the significant difference between two system Spans of the "Frontend Service" indicates that the root cause of the slow response lies in the processing process of the "Frontend Service". - -![Slow Call Flame Graph Example 2 - Significant Difference Between System Spans](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d292867f8b.png) - -- **Example 3: Significant Length of Terminal System Span** - -In the figure below, the significant length of the "DNS Service" indicates that the root cause of the slow response lies in the processing process of the "DNS Service". - -![Slow Call Flame Graph Example 3 - Significant Length of Terminal System Span](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d292a82a96.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/04-line.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/04-line.md deleted file mode 100644 index 83d2ba94..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/04-line.md +++ /dev/null @@ -1,93 +0,0 @@ ---- -title: Line Chart -permalink: /guide/ee-tenant/dashboard/panel/line/ ---- - -> This document was translated by ChatGPT - -# Line Chart - -Line charts display continuous data over time, making them ideal for viewing trends within a specific time range. - -In DeepFlow, line charts are categorized into two types: Standard Line Chart and TOP N Line Chart. - -- Standard Line Chart: Displays the changes in all queried data over time. -- TOP N Line Chart: First, the queried data is grouped and the top N are selected, then the changes in the data of these top N services or resources over time are displayed. - -## Overview - -![00-总览](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664b0fd1d1069.png) - -- **① Query Area:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Modify Metrics](./topology/). -- **② Modify Metrics:** Basic operations of the chart. For details, please refer to the section [Traffic Topology](./topology/). -- **③ Settings:** Basic operations of the chart. For details, please refer to the section [Settings]. -- **④ Delete:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Overview](./topology/). - -### Settings - -Users can click the `Settings` button to operate the chart. - -![01-设置](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240415661cc68097aa9.png) - -- **Edit:** Allows modification and editing of the chart, such as changing search conditions, names, saving view positions, and opening the original function page of the chart. For details, please refer to the section [Edit]. -- **Copy:** A capability within the `Dashboard`, supporting the copying of charts within the dashboard. -- **Download CSV Data:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Settings](./topology/). -- **View API:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Settings](./topology/). - -### Edit - -The line chart editing box consists of three parts: `① Chart`, `② Search Conditions`, and `③ Configuration`. - -![02-编辑.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664ac8cae92be.png) - -- **① Chart:** The chart is drawn based on `② Search Conditions` and `③ Configuration`. -- **② Search Conditions:** For the usage of search conditions, please refer to the section [Search](../../query/overview/). -- **③ Configuration:** Supports quick switching of chart types, and configuration of chart styles and related functions. - - **Switch Chart Type:** Allows quick switching of chart types, only supported for charts within the dashboard. - - **General Configuration:** Rich functionalities to set the chart style. - - **Chart Information:** Supports editing the chart name and adding descriptions. - - Title: Supports modifying the chart name. - - Description: Supports adding related description information to the chart in markdown format, allowing links, images, etc. - - **Metric Settings:** Supports setting aliases, units, and thresholds for the metrics added in `② Search Conditions`. - - **Chart Style:** Configuration of the display style of the line chart. - - Display Form: Can choose from `Line`, `Bar`, or `Point` to draw the chart. - - Drawing Method: Can choose from four methods to connect the `Line`. - - Node Display: Data nodes can be displayed as hollow circles, hollow squares, solid circles, or solid squares. - - Line Style: Data lines can be displayed as solid, dashed, or dotted lines. - - Area Fill: The area between the line and the coordinate axis can be filled with color. - - Display Values: Shows the corresponding data series values as the mouse moves. - - Data Stacking: Supports displaying multiple data series in a stacked or tiled manner. - - Stacked: Displays the values of multiple data series stacked in the same coordinate system to show their overall trend and individual contributions. - - Tiled: Displays the values of multiple data series tiled in the same coordinate system to better show their differences and relationships. - - **Color:** Supports setting the color of the current chart. - - **Legend:** Sets the display status, position, values, and display form of the legend. - - Mode: Can choose to display as a legend or table. - - Position: Supports displaying below or to the right of the chart. - - Display Values: Can choose to display the `Avg, Max, Min, Max` values of the metrics. - - **Axis Lines:** Sets the background lines and coordinate axis lines. - - Background Lines: Supports setting straight lines, grid lines, or turning off background lines. - - Coordinate Axis Display: Can choose to show or hide the coordinate axis. - - **Data Filtering:** Can choose to hide data with null or zero values. - - **Advanced Configuration:** - - **Tip:** Configuration of the tip display method. - - Tip Mode: Can choose from `All`, `Single`, or `Hide`. - - All: Displays data for all series. - - Single: Only displays data for the series highlighted by the mouse. - - Hide: Does not display the tip. - - Series Name: Can choose to show or hide the series name. - - **Data Filtering:** - - Value Filling Method: Supports three methods to fill `null data`. - - Fill 0: Default data filling method. - - Line Chart: Fills the line chart data as `0`, and the tip displays the value as `0`. - - Bar Chart: Does not display height in the bar chart, and the bar chart tip displays the data as `0`. - - Fill null: No value at the time point. - - Line Chart: The line chart is interrupted, and the tip displays the value as `null`. - - Bar Chart: Does not display height in the bar chart, and the bar chart tip displays the data as `null`. - - Fill none: No time point. - - Line Chart: Ignores this point and connects to the next point, and the tip for this time point displays the tip of the previous time point with data. - - Bar Chart: Does not display height in the bar chart, and the tip for this time point displays the tip of the previous time point with data. - - Hide Series: Supports hiding `series with all data as 0` or `series with all data as null`. - - **Data Sorting:** - - Top Sorting: Supports sorting data in ascending/descending order. - - Top N: Combined with `Top Sorting`, returns the top few data with the smallest/largest values. - - Can choose `Top 5`, `Top 10`, `Top 20`. diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/05-bar.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/05-bar.md deleted file mode 100644 index b0ae3441..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/05-bar.md +++ /dev/null @@ -1,31 +0,0 @@ ---- -title: Bar Chart -permalink: /guide/ee-tenant/dashboard/panel/bar/ ---- - -> This document was translated by ChatGPT - -# Bar Chart - -A bar chart represents data for various categories by plotting a series of vertical or horizontal bars, providing an intuitive display of data size and distribution. In DeepFlow, data can be visualized as a `bar chart` by switching the display mode in the `table`. - -## Overview - -![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f8001c6c54e.png) - -- **① Switch Query Area:** Basic chart operation. For details, please refer to the section [Traffic Topology - Overview](./topology/) -- **② Switch Top Data:** Basic chart operation. For details, please refer to the section [Traffic Topology - Overview](./topology/) -- **③ Modify Metrics:** Basic chart operation. For details, please refer to the section [Traffic Topology - Modify Metrics](./topology/) -- **④ Settings:** Standard chart operation. For details, please refer to the section [Line Chart - Settings](./line/) -- **⑤ Delete:** A capability within the `Dashboard`. For details, please refer to the section [Traffic Topology - Overview](./topology/) - -### Bar Chart - -The bar chart editing box consists of three parts: `① Chart`, `② Search Conditions`, and `③ Style and Settings`. - -![01-Bar Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f8001b3666e.png) - -- **① Chart:** The chart is drawn based on `② Search Conditions` and `③ Style and Settings` -- **② Search Conditions:** For the usage of search conditions, please refer to the section [Search](../../query/overview/) -- **③ Style and Settings:** For setting the style of the chart, please refer to the section [Line Chart - Edit](./line/) - - **Top Sorting:** Supports ascending/descending sorting of data \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/06-pie.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/06-pie.md deleted file mode 100644 index eb99f23a..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/06-pie.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -title: Pie Chart -permalink: /guide/ee-tenant/dashboard/panel/pie/ ---- - -> This document was translated by ChatGPT - -# Pie Chart - -A pie chart visually represents the relative size and proportion of data for different categories by dividing the entire circle into multiple sectors. In DeepFlow, data can be visualized in the form of a `pie chart` by switching the display mode in the `table`. - -![Pie Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509754fce717.png) - -- **① Switch Top Data:** Basic chart operation. For details, please refer to the section [Traffic Topology - Overview Introduction](./topology/) -- **② Modify Metrics:** Basic chart operation. For details, please refer to the section [Traffic Topology - Modify Metrics](./topology/) -- **③ Settings:** Standard chart operation. For details, please refer to the section [Line Chart - Settings](./line/) -- **④ Delete:** A capability in the `Dashboard`. For details, please refer to the section [Traffic Topology - Overview Introduction](./topology/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/07-histogram.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/07-histogram.md deleted file mode 100644 index 4b396149..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/07-histogram.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -title: Histogram -permalink: /guide/ee-tenant/dashboard/panel/histogram/ ---- - -> This document was translated by ChatGPT - -# Histogram - -A histogram represents the distribution of data by dividing it into several continuous intervals (also known as "bins" or "buckets") and plotting the frequency (or count) of data within each interval. Histograms help to visually display the central tendency, dispersion, and shape of the data. - -![Histogram](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230919650975509aeb6.png) - -- **① Query Area:** Supports switching between different query areas. The query area for the histogram only supports `single selection`. -- **② Modify Metric:** Basic operation of the chart. For details, please refer to the section [Traffic Topology - Modify Metric](./topology/). -- **③ Settings:** Standard operation of the chart. For details, please refer to the section [Traffic Topology - Settings](./topology/). -- **④ Delete:** A capability within the `Dashboard`. For details, please refer to the section [Traffic Topology - Overview](./topology/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/08-table.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/08-table.md deleted file mode 100644 index b9ec3e33..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/08-table.md +++ /dev/null @@ -1,80 +0,0 @@ ---- -title: Table -permalink: /guide/ee-tenant/dashboard/panel/table/ ---- - -> This document was translated by ChatGPT - -# Table - -Tables are used to display detailed information of structured data. DeepFlow tables can be divided into two types: `Aggregate Table` and `Detail Table`. - -## Aggregate Table - -Aggregate tables support querying data from multiple tables of the same type simultaneously, such as `service metrics`, `path metrics`, or `xx logs`. - -![01-Aggregate Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031965f8f90497a37.png) - -- **① Query Area:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Modify Metrics](./topology/) -- **② Modify Metrics:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Overview](./topology/) - - Long press and drag the data to map the sorting to the table -- **③ Settings:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Settings](./topology/) -- **④ Delete:** A capability within the `Dashboard`. For details, please refer to the section [Traffic Topology - Overview](./topology/) - -### Edit - -The edit box of the aggregate table consists of three parts: `① Chart`, `② Search Conditions`, and `③ Configuration`. - -![02-Edit](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664aff296a339.png) - -- **① Chart:** The chart is drawn based on `② Search Conditions` and `③ Configuration` -- **② Search Conditions:** For the usage of search conditions, please refer to the section [Search](../../query/overview/) -- **③ Configuration:** Supports quick switching of chart types, and configuration of chart styles and related functions - - **Switch Chart Type:** Basic functionality of the chart. For details, please refer to the section [Line Chart](./line/) - - **Common Configurations:** Rich functionalities to set the chart style - - **Chart Information:** Basic functionality of the chart. For details, please refer to the section [Line Chart](./line/) - - **Color:** Set the basic color for the text or background of the chart - - Note: Only effective for columns with `Column Settings - Color` enabled - - **Column Settings:** Supports setting the color, alignment, and value display of columns - - Color: Configured colors can be selected for coloring objects, with options for no effect, text, or background - - Column Alignment: Choose the alignment position of the column, with options for left, center, or right - - Value Mapping: Three methods to match specified column values and replace them with custom text content - - Text: Match through strings - - Range: Match through numerical ranges - - Regular Expression: Match through regular expressions - - Note: The priority of value mapping effectiveness is `Text > Range = Regular Expression`. When there are matching conditions of the same priority, the one higher in the order takes effect - - Threshold: Set the numerical range, and the text/background within the specified range will display the specified color - - Unit: Set the unit of the metric - - Alias: Set the alias of the metric - - **Advanced Configuration:** - - **Cell:** Supports configuration of the table copy function - - Copy Function: Enable or disable the table content copy function - - Copy Content: Choose the data content to copy - - Copy Data: Only copy the data content of the current cell, i.e., `value` - - Forward Filter Condition: The copy content format is `key: value`, which can be pasted into the search bar on the page. The search bar can quickly recognize it as a `search tag` for querying - - Forward Filter Condition: The copy content format is `key!: value`, which can be pasted into the search bar on the page. The search bar can quickly recognize it as a `search tag` for querying - - For details on using search tags, please refer to the section [Service Search Box](../../query/service-search/) - - **Table Settings:** Supports setting the border and header of the table - -## Detail Table - -Detail tables only support querying a single type of log data, such as flow logs or call logs. - -![03-Detail Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031965f8f7906c908.png) - -- **① Query Area:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Modify Metrics](./topology/) -- **② Column Selection:** Supports searching, adding, and deleting column options supported by the current table - - Long press and drag the data to map the sorting to the table -- **③ Settings:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Settings](./topology/) -- **④ Delete:** A capability within the `Dashboard`. For details, please refer to the section [Traffic Topology - Overview](./topology/) - -### Edit - -The edit box of the detail table consists of three parts: `① Chart`, `② Search Conditions`, and `③ Configuration`. - -![04-Edit](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664aff2835cce.png) - -- **① Chart:** The chart is drawn based on `② Search Conditions` and `③ Configuration` -- **② Search Conditions:** For the usage of search conditions, please refer to the section [Search](../../query/overview/) - > Note: Detail tables do not support adding multiple query conditions -- **③ Configuration:** For details, please refer to the section [Aggregate Table - Edit] \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/09-stat.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/09-stat.md deleted file mode 100644 index bf62c334..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/09-stat.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -title: Overview Chart -permalink: /guide/ee-tenant/dashboard/panel/stat/ ---- - -> This document was translated by ChatGPT - -# Overview Chart - -The overview chart displays statistical values based on query conditions. - -## Overview Introduction - -![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7e22138b9b.png) - -- **① Query Area:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Overview Introduction](./topology/) -- **② Modify Metrics:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Modify Metrics](./topology/) -- **③ Settings:** Basic operations of the chart. For details, please refer to the section [Settings] -- **④ Delete:** Basic operations of the chart. For details, please refer to the section [Traffic Topology - Overview Introduction](./topology/) - -### Overview Chart - -The overview chart editing box consists of three parts: `① Chart`, `② Search Conditions`, and `③ Style and Settings`. - -![01-Overview Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7e4051ebe2.png) - -- **① Chart:** The chart is drawn based on `② Search Conditions` and `③ Style and Settings` -- **② Search Conditions:** For the usage of search conditions, please refer to the section [Search](../../query/overview/) -- **③ Style and Settings:** Set the style, color, etc., of the chart - - **Style:** Rich functionalities to support style settings for the chart - - **Title:** Supports modifying the chart name - - **Unit:** Supports editing the unit. If no unit is set, the default unit of the metric will be used - - **Data Precision:** Supports setting the decimal places for displaying data - - 1 decimal place, 2 decimal places, 3 decimal places, integer, full precision (displays all decimals of the metric) - - **Color:** Supports setting the color of the font, background, and background image - - **Settings:** - - **Background Image:** Supports displaying a line chart or bar chart of the metric over a time period - - **Comparison:** Supports displaying a comparison of the metric values between the current time and one hour ago, or one day ago \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/10-text.md b/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/10-text.md deleted file mode 100644 index 39ebd8bd..00000000 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/99-panel/10-text.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -title: Text -permalink: /guide/ee-tenant/dashboard/panel/text/ ---- - -> This document was translated by ChatGPT - -# Text - -Text is often used for overall description and tips in the dashboard. - -## Overview Introduction - -![总览.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466431ce726d72.png) - -- Supports markdown text format -- Supports adding images, hyperlinks, etc. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/03-universal-map/01-overview.md b/translate/translated/06-guide/01-ee-tenant/03-universal-map/01-overview.md deleted file mode 100644 index a3423286..00000000 --- a/translate/translated/06-guide/01-ee-tenant/03-universal-map/01-overview.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Overview -permalink: /guide/ee-tenant/universal-map/overview/ ---- - -> This document was translated by ChatGPT - -# Overview - -The universal map allows users to independently define each business and the services or service groups within each business, enabling users to construct their service topology in a more scenario-based manner. - -DeepFlow's universal map is divided into three main pages, which will be detailed in the following sections. - -- [Business Definition](./business-def/) -- [Service List](./service-list/) -- [Service Topology](./service-map/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/03-universal-map/02-business-def.md b/translate/translated/06-guide/01-ee-tenant/03-universal-map/02-business-def.md deleted file mode 100644 index 25eadd28..00000000 --- a/translate/translated/06-guide/01-ee-tenant/03-universal-map/02-business-def.md +++ /dev/null @@ -1,112 +0,0 @@ ---- -title: Business Definition -permalink: /guide/ee-tenant/universal-map/business-def/ ---- - -> This document was translated by ChatGPT - -# Business Definition - -Business definition includes the business name, the definition of data tables, as well as the definition of service groups, services, and paths within the business. Below is a detailed introduction on how to define them. - -![00-Term Explanation](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f64f1d682.jpg) - -A business can consist of several `services` and `paths`. `Services` can be fully customized by the user and added to `custom service groups`; alternatively, service groups can be set up to automatically generate `services` within the group, corresponding to `auto-grouped service groups`. `Paths` are defined by specifying the access relationships of services. The following explains the above diagram: - -- Business: Mobile Banking Business -- Services: - - Independent services not added to any group: Frontend Load Service - - Added to custom service group, Frontend Service Group: Operations Frontend Service, Financial Frontend Service - - Services generated by auto-grouped service group, Midend Service Group: Rights Center Service, Search Center Service, Average Center Service, Payment Center Service -- Paths: - - Custom: Frontend Load Service -> Operations Frontend Service; Frontend Load Service -> Financial Frontend Service; Operations Frontend Service -> Midend Service Group (auto-grouped); Financial Frontend Service -> Midend Service Group (auto-grouped) - - Auto-generated: Access relationships between services within the Midend Service Group (auto-grouped) - -## Business List - -![01-Business List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9c6679b3.png) - -- **① Create New Business**: Supports creating a new business. For details, please refer to the [Create New Business] section -- **② Name**: Click to enter the `Business Details Page`. For details, please refer to the [Business Details Page] section -- **③ Star**: Click to `star` or `unstar`, the list will be sorted by default starred + name dictionary order in descending order -- **④ Service Topology**: Click to jump to the `Service Topology` page to view the current business in a waterfall topology -- **⑤ Service List**: Click to jump to the `Service List` page -- **⑥ Edit**: Edit the business -- **⑦ Delete**: Delete the business - -### Create New Business - -![02-Create New Business](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e3d393f6.png) - -- Name: Required, business name -- Data Table: Required, the data table from which the business data originates - - To view the network metrics of the business, select `Network-Path-Metric Data (xx)`. To view the application metrics of the business, select `Application-Path-Metric Data (xx)` -- Metric Quantity: Depending on the data table, the corresponding metric quantity can be used - - Supports setting up to 10 metric quantities - -### Business Details Page - -The business details page consists of `Basic Information and Operations` and `Details List`, where you can define `services`, `service groups`, and `paths` for the current business. - -#### Services - -![03-Services](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e3e6a56d.png) - -- Basic Information and Operations - - Basic Information: Displays the data table, metrics, and the number of services, service groups, and paths for the business - - Click on the statistics to quickly switch the list below to the corresponding `services`, `service groups`, or `paths` - - Operations - - Edit: Supports modifying the name, data table, and metrics of the current business. For details, please refer to the [Create New Business] section - - Star: After setting the star, it will be displayed first on the `Business List` page - - Service Topology: Jump to the `Service Topology` page to view the topology of the business. For details, please refer to the [Service Topology](./service-map/) section - - Service List: Jump to the `Service List` page to view the topology of the business. For details, refer to the [Service Topology](./service-list/) section -- Service List: Information related to all services in the current business, supports editing and deleting - - For example, Redis/DNS/Access Client/Stress Test Client are all independent services - -![04-Create New Service](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e3fcabc4.png) - -- Name: Required, service name must be unique within the same business -- Icon: The service ICON displayed in `Service Topology` and `Service List`, currently selectable as `Resource-Icon` -- Filter Conditions: Define service data based on filter conditions, supports `bi-directional/uni-directional` queries - - Independently set client and server: If checked, use the client and server filter condition boxes for `uni-directional` filtering; if unchecked, perform `bi-directional` filtering - - Direction: Define filter conditions for the service in different roles - - Bi-directional: A single filter condition box for both client and server filter conditions - - Uni-directional: `Server` and `Client` two filter condition boxes - - Server: Filter conditions only as a server - - Client: Filter conditions only as a client - - For filter condition operations, please refer to the [Query](../query/overview/) section - - Group: Default is `*` -- Service Group: Supports adding to `custom type service groups`, a `service` can only be added to one `service group`. For the definition of `service groups`, refer to subsequent sections -- Metric Threshold: Can adjust the metric threshold for each service - - By default, metrics are collapsed, click the expand button to expand and edit - - When the metric quantity exceeds the threshold, the corresponding `service` display will be highlighted in red - -#### Service Groups - -![05-Service Groups](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e41e9e32.png) - -- Service Group List: Each row represents a group of services within the business, such as the Frontend Service Group, which can include APP Access Service, Web Access Service. DeepFlow service groups can be user-defined, adding custom services one by one, or automatically recognized groups of services can form service groups. - -![06-Create New Service Group](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e4586d8e.png) - -- Name: Required, service group name must be unique within the same business -- Type: Divided into `auto-grouped` and `custom` types - - Custom: User-defined, can independently select `services` to join - - Auto-grouped: A service group automatically recognized based on the `filter conditions` set - - Group: Can be grouped according to auto_service or [custom auto-grouping tags](../../../features/auto-tagging/custom-tags) - - For `Direction` and `Filter Conditions` usage instructions, please refer to the [Services] section -- Metric Threshold - -#### Paths - -![07-Paths](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e4869fe5.png) - -- Path List: Each row represents a path where a `service`/`service group` accesses another `service`/`service group`. For example, `Client=Service A, Server=Service B` indicates that `Service A` accesses `Service B` based on the `client` filter conditions and the `server` filter conditions to query the path data. - - Batch Delete: Check the checkbox to support batch deletion - -![08-Create New Path](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e4a3bd2a.png) - -- Name: Required, path name -- Client: Multi-select, options are `All`, `custom services and auto-grouped` types of service groups -- Server: Multi-select, options are `All`, `custom services and auto-grouped` types of service groups -- All: Indicates selecting all custom services/auto-grouped types of service groups \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/03-universal-map/03-service-list.md b/translate/translated/06-guide/01-ee-tenant/03-universal-map/03-service-list.md deleted file mode 100644 index 393d2b0b..00000000 --- a/translate/translated/06-guide/01-ee-tenant/03-universal-map/03-service-list.md +++ /dev/null @@ -1,23 +0,0 @@ ---- -title: Service List -permalink: /guide/ee-tenant/universal-map/service-list/ ---- - -> This document was translated by ChatGPT - -# Service List - -View the metric data of each service in the business in a list format - -![01-Service List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240407661262d40a411.png) - -- **Business Switch Dropdown**: Quickly switch between businesses, defaulting to the first starred business -- **Modify Metrics**: Support displaying/hiding metrics in the table - - Set Primary Metric: The set primary metric will be displayed before other metrics -- **Settings**: Support `View API`, `Service Management`, etc. -- **Table**: - - Name: Service name, ICON represents the service type - - Service Group: The service group to which the service belongs - - Region: The region to which the service belongs - - Metrics: Statistics of the service as a server, will be marked in red when exceeding the threshold - - Actions: Double-click the `table row` to enter the right slide panel to view detailed information of the service. For details on using the right slide panel, please refer to the chapter [Service Topology - Right Slide Panel](./service-map/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/03-universal-map/04-service-map.md b/translate/translated/06-guide/01-ee-tenant/03-universal-map/04-service-map.md deleted file mode 100644 index 3cde96df..00000000 --- a/translate/translated/06-guide/01-ee-tenant/03-universal-map/04-service-map.md +++ /dev/null @@ -1,97 +0,0 @@ ---- -title: Service Topology -permalink: /guide/ee-tenant/universal-map/service-map/ ---- - -> This document was translated by ChatGPT - -# Service Topology - -Displays the `services` defined by the user in `Business Definition` in a waterfall topology format. - -![01-Service Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9bb16811.png) - -- **① Business Switch Dropdown**: Quickly switch businesses, defaults to the first starred business -- **② Service Management**: Click the button to enter the business details page -- **③ Modify Metrics**: Supports displaying/hiding metrics and setting primary metrics -- **④ Save**: Supports saving the `time range`, `service topology position`, and `configuration` of the service topology -- **⑤ Settings**: Supports `edit`, `view API`, `add to Dashboard`, `reset`, and other functions - - Reset: The service topology will revert to its initial layout -- **⑥ Service Group**: Consists of `name row` + `box`, as shown in the figure, client, gcp-microservices-demo, DNS, and Redis are all independent service groups -- **⑦ Path**: Represents the actual data path from the client to the server, hover to view TIP, click to view path details through the right sliding panel, see subsequent chapters for details -- **⑧ Service**: Each block in the topology represents a service, consisting of `name row` + `metrics`, hover to view TIP, click to view service details through the right sliding panel, see subsequent chapters for details - - Name Row: ICON represents the service type - - Metrics: Displays metrics according to the priority of [observability points](../../../features/universal-map/auto-metrics) (s-xx > local > rest > app). When the metrics exceed the threshold, the name row and corresponding metrics will be marked in red -- **⑨ Operation Set**: Supports layout, connection, zooming, and other operations on the topology - - Layout: Enter manual layout mode, supports dragging `services`, click the `save` button again to exit and save the layout position - - Edit: Enter path editing mode, supports adding or deleting `paths`. Click the `edit` button again to exit the path editing mode - - Add Path: Supports adding connections between `services/auto-grouped service groups`, converting connections to `paths` - - Delete Path: Click the close button on the path to delete the corresponding `path` - - Zoom In/Out: Zoom in or out of the topology - - Scroll Wheel Zoom: When scroll wheel zoom is disabled, only the `zoom in/out` buttons can be used to control the size of the topology - -## Right Sliding Panel - -Clicking on `service` or `path` will enter the right sliding panel to view detailed information. The right sliding panel consists of the upper `call topology` and the lower TAB. - -### Call Topology - -View the client and server of the selected service through `call topology`, and also view the selected path. - -![02-Call Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9aa20975.png) - -- **① Switch Group**: View the call topology by \*, auto_service, auto_instance, and [custom auto-grouping tags](../../../features/auto-tagging/custom-tags) -- **② Name**: The name of the currently clicked `service` or `path`, corresponding to the object viewed in the lower TAB. -- **③ Node**: Refer to [Traffic Topology](../dashboard/panel/topology/) introduction -- **④ Path**: Refer to [Traffic Topology](../dashboard/panel/topology/) introduction - -### Knowledge Graph - -![03-Knowledge Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f435c6d.png) - -Refer to [Application - Right Sliding Panel - Knowledge Graph](../tracing/right-sliding-box/) introduction - -### Application Performance - -Use `Application Performance` to analyze whether there are application layer anomalies in the selected service or path. - -![04-Application Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f6ac6b5.png) - -The TAB consists of three curve charts for throughput, latency, and anomalies, and a list of endpoints below. Clicking on a row in the endpoint list will enter the next level of the right sliding panel. - -![04-1-Application Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f764e5d.png) -![04-2-Application Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f799476.png) - -The next level of the right sliding panel can view the RED metrics and log details of a specific endpoint. When there are anomalies, `anomaly analysis` can be viewed. - -### Network Performance - -Use `Network Performance` to analyze whether there are application layer anomalies in the selected service or path. - -![05-Network Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f9625df.png) - -The TAB consists of four curve charts for throughput, latency, anomalies, and performance, and a list of service ports below. Clicking on a row in the list will enter the next level of the right sliding panel. - -- Service: View data for the service `as a client` or `as a server` separately -- Path: Click on each `observability point` in the `topology analysis` to view the data of each `observability point` separately - -![05-1-Network Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f9c515d.png) -![05-2-Network Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3fd02700.png) - -The next level of the right sliding panel can view the RED metrics and log details of a specific service port. When there are anomalies, `anomaly analysis` can be viewed. - -### Infrastructure - -Use `Infrastructure` to analyze the CPU, memory, status, and other data of the infrastructure instances corresponding to the service. - -![06-Infrastructure](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9ae67608.png) - -**① Switch Service:** Switch the service to view the infrastructure, the options are the services at both ends of the clicked path or the clicked service - -### Events - -View resource change events of `service` or `path` - -![06-Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3fcdb8b4.png) - -Refer to [Tracing - Right Sliding Panel - Events](../tracing/right-sliding-box/) introduction \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/01-overview.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/01-overview.md deleted file mode 100644 index 6b406fab..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/01-overview.md +++ /dev/null @@ -1,20 +0,0 @@ ---- -title: Overview -permalink: /guide/ee-tenant/tracing/overview/ ---- - -> This document was translated by ChatGPT - -# Overview - -The DeepFlow application module supports real-time monitoring of service golden metrics, presenting service call topology, in-depth analysis of service call logs, and initiating blind-spot-free distributed tracing. By adopting a zero-intrusion approach to applications, it enables observability for application services, allowing users to efficiently identify and locate performance bottlenecks at the application layer, quickly discover and resolve errors and anomalies in application services, and perform targeted optimizations. This significantly enhances the performance and reliability of application services. - -DeepFlow's application is divided into six main pages, which will be detailed in the following sections. - -- [Resource Analysis](./service-list/) -- [Path Analysis](./service-statistics/) -- [Topology Analysis](./path-topology/) -- [Call Logs](./call-log/) -- [Distributed Tracing](./call-chain-tracing/) -- [File Reading and Writing](./file-reading-and-writing/) -- [Right Sliding Box](./right-sliding-box/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/02-service-list.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/02-service-list.md deleted file mode 100644 index 95ef08ff..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/02-service-list.md +++ /dev/null @@ -1,32 +0,0 @@ ---- -title: Resource Analysis -permalink: /guide/ee-tenant/tracing/service-list/ ---- - -> This document was translated by ChatGPT - -# Resource Analysis - -The Resource Analysis page provides a centralized way to present an overview of the application services monitored by DeepFlow. It includes information such as the names of various application services, sources, and golden metrics. Through the Resource Analysis page, users can quickly obtain the overall status of application services, easily locate the services that need attention, and further view detailed information for performance analysis, fault diagnosis, and optimization. This helps improve the efficiency and accuracy of service monitoring. - -## Overview Introduction - -The Resource Analysis page supports time filter queries, conditional search queries, and other methods for application service overview queries. - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a602e67679.png) - -- **① Time Selector**: Supports time filter queries. For details, please refer to the chapter [Dashboard Details - Time Selector](../dashboard/use/) -- **② Search Snapshot**: Supports saving search conditions as snapshots. For details, please refer to the chapter [Query - Search Snapshot](../query/history/) -- **③ Search Save and Settings**: - - Search Save: Supports quickly saving the search conditions of the current page. For details, please refer to the chapter [Query - Search Snapshot](../query/history/) - - Settings: A collection of page setting operations - - Database Fields: Supports viewing the tags and metrics in the data table used on the current page - - Enable/Disable Tip Sync: When enabled, you can view the metric data of all line charts at the same time point - - Switch Interpolation Method: When data does not exist at a certain time point, you can switch the interpolation method to handle it as needed - - Switch Stacking: Quickly switch the display form of all time-series related Panels on the functional page between tiled/stacked - - Default/Full Name Display: Display the full name or default name of the legend -- **④ Search Box**: Supports searching or grouping by Tag. For details, please refer to the chapter [Query](../query/overview/) -- **⑤ Left Quick Filter**: Allows quick data filtering. For details, please refer to the chapter [Query - Left Quick Filter](../query/left-quick-filter/) -- **⑥ Area Query**: Supports quickly switching query area data -- **Table Operations**: - - Click Row: Clicking on the data in the table allows you to quickly enter the right sliding box to view the relevant information of the corresponding application service. For details, please refer to the chapter [Right Sliding Box](./right-sliding-box/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/03-service-statistics.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/03-service-statistics.md deleted file mode 100644 index b0cc45be..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/03-service-statistics.md +++ /dev/null @@ -1,30 +0,0 @@ ---- -title: Path Analysis -permalink: /guide/ee-tenant/tracing/service-statistics/ ---- - -> This document was translated by ChatGPT - -# Path Analysis - -The Path Analysis page, based on the Resource Analysis page, displays the client and server of the request application. It allows for more flexible analysis of application performance metrics from multiple dimensions, providing insights into service request rates, response times, and error ratios, which helps in identifying system bottlenecks and optimizing system performance. - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a6ba9e6900.png) - -- **① Time Selector**: Supports time filter queries. For details, please refer to the chapter [Dashboard Details - Time Selector](../dashboard/use/). -- **② Search Snapshot**: Supports saving search conditions as snapshots. For details, please refer to the chapter [Query - Search Snapshot](../query/history/). -- **③ Search Save and Settings**: - - Search Save: Supports quickly saving the search conditions of the current page. For details, please refer to the chapter [Query - Search Snapshot](../query/history/). - - Settings: A collection of page setting operations. - - Database Fields: Supports viewing the tags and metrics in the data table used on the current page. - - Enable/Disable Tip Sync: When enabled, you can view the metric data of all line charts at the same time point. - - Switch Interpolation Method: When data does not exist at a certain time point, you can switch the interpolation method to handle it as needed. - - Switch Stacking: Quickly switch the display form of all time-series related Panels on the functional page between tiled/stacked. - - Name Default/Full Display: The legend's name can be displayed in full or default. -- **④ Search Box**: Supports searching or grouping by Tag. For details, please refer to the chapter [Query](../query/overview/). -- **⑤ Left Quick Filter**: Allows quick data filtering. For details, please refer to the chapter [Query - Left Quick Filter](../query/left-quick-filter/). -- **⑥ Area Query**: Supports quickly switching the query area data. -- **Table Operations**: - - Click Row: Clicking on the data in the table allows you to quickly enter the right sliding box to view the relevant information of the corresponding application service. For details, please refer to the chapter [Right Sliding Box](./right-sliding-box/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/04-path-topology.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/04-path-topology.md deleted file mode 100644 index e8f9aafb..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/04-path-topology.md +++ /dev/null @@ -1,26 +0,0 @@ ---- -title: Topology Analysis -permalink: /guide/ee-tenant/tracing/path-topology/ ---- - -> This document was translated by ChatGPT - -# Topology Analysis - -The topology analysis page displays the dependencies between services or resources in the form of a topology. By combining threshold values of metrics, you can quickly identify bottlenecks and issues within the system and take timely actions to respond and address them. Additionally, by continuously monitoring and updating the topology analysis path, you can continuously optimize system architecture and performance. - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a6d7f5039d.png) - -- **① Time Selector**: Supports time filter queries. For usage details, please refer to the chapter [Dashboard Details - Time Selector](../dashboard/use/) -- **② Search Snapshot**: Supports saving search conditions as snapshots. For usage details, please refer to the chapter [Query - Search Snapshot](../query/history/) -- **③ Search Save and Settings**: - - Search Save: Supports quickly saving the search conditions of the current page. For usage details, please refer to the chapter [Query - Search Snapshot](../query/history/) - - Settings: A collection of page setting operations - - Database Fields: Supports viewing the tags and metrics in the data table used on the current page - - Name Default/Full Display: Toggle between displaying the full name or default name of the legend -- **④ Search Box**: Supports searching or grouping by Tag. For usage details, please refer to the chapter [Query](../query/overview/) -- **⑤ Left Quick Filter**: Allows quick data filtering. For usage details, please refer to the chapter [Query - Left Quick Filter](../query/left-quick-filter/) -- **⑥ Area Query**: Supports quickly switching query area data -- **Topology Graph Operations**: Double-click on data nodes to enter the right slide panel to view corresponding information. For usage details, please refer to the chapter [Traffic Topology](../dashboard/panel/topology/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/05-call-log.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/05-call-log.md deleted file mode 100644 index 98b31d9e..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/05-call-log.md +++ /dev/null @@ -1,28 +0,0 @@ ---- -title: Call Log -permalink: /guide/ee-tenant/tracing/call-log/ ---- - -> This document was translated by ChatGPT - -# Call Log - -The call log records detailed information for each call. - -## Overview - -![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cbbf9e6ac2.png) - -- **① Time Selector**: Supports time filter queries. For details, please refer to the section [Dashboard Details - Time Selector](../dashboard/use/) -- **② Search Snapshot**: Supports saving search conditions as snapshots. For details, please refer to the section [Query - Search Snapshot](../query/history/) -- **③ Search Save and Settings**: - - Search Save: Supports quickly saving the search conditions of the current page. For details, please refer to the section [Query - Search Snapshot](../query/history/) - - Settings: A collection of page setting operations - - Database Fields: Supports viewing the tags and metrics in the data table used on the current page - - Name Default/Full Display: Toggle between displaying the full name or default name of the legend -- **④ Service Search Box**: Supports searching or grouping by Tag. For details, please refer to the section [Query](../query/overview/) -- **⑤ Left Quick Filter**: Allows quick data filtering. For details, please refer to the section [Query - Left Quick Filter](../query/left-quick-filter/) -- **⑥ Area Query**: Supports quickly switching query area data -- **Table Operations**: - - Click Row: Clicking on the table data allows you to quickly enter the right sliding box to view related information of the corresponding application service. For details, please refer to the section [Right Sliding Box](./right-sliding-box/) - - ⑦ Copy: Table content copy function. After clicking, you can select the content format to copy. For configuration details, please refer to the section [Table](../dashboard/panel/table/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/06-call-chain-tracing.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/06-call-chain-tracing.md deleted file mode 100644 index 0372e2cb..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/06-call-chain-tracing.md +++ /dev/null @@ -1,27 +0,0 @@ ---- -title: Distributed Tracing -permalink: /guide/ee-tenant/tracing/call-chain-tracing/ ---- - -> This document was translated by ChatGPT - -# Distributed Tracing - -Distributed tracing records detailed information for each call, supporting only data collected via eBPF or calls initiated through the OpenTelemetry protocol transmitted to DeepFlow. - -## Overview - -![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566442db5b7387.png) - -- **① Time Selector**: Supports time filter queries. For details, please refer to the chapter [Dashboard Details - Time Selector](../dashboard/use/) -- **② Search Snapshot**: Supports saving search conditions as snapshots. For details, please refer to the chapter [Query - Search Snapshot](../query/history/) -- **③ Search Save and Settings**: - - Search Save: Supports quickly saving the search conditions of the current page. For details, please refer to the chapter [Query - Search Snapshot](../query/history/) - - Settings: A collection of page setting operations - - Database Fields: Supports viewing tags and metrics in the data table used on the current page - - Name Default/Full Display: Full or default display of legend names -- **④ Service Search Box**: Supports searching or grouping by Tag. For details, please refer to the chapter [Query](../query/overview/) -- **⑤ Left Quick Filter**: Allows quick data filtering. For details, please refer to the chapter [Query - Left Quick Filter](../query/left-quick-filter/) -- **⑥ Area Query**: Supports quick switching of query area data -- **Application Tracing Table**: Displays call information between services or resources within a certain period, such as client, server, request resource, request type, request domain, etc. For details, please refer to the chapter [Table](../dashboard/panel/table/) - - **Operation:** Click on a list row to enter the right sliding box to view the entire call lifecycle traced from the request. For details, please refer to the chapter [Right Sliding Box - Distributed Tracing](./right-sliding-box/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/07-file-reading-and-writing.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/07-file-reading-and-writing.md deleted file mode 100644 index f5d2bb39..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/07-file-reading-and-writing.md +++ /dev/null @@ -1,14 +0,0 @@ ---- -title: File Reading and Writing -permalink: /guide/ee-tenant/tracing/file-reading-and-writing/ ---- - -> This document was translated by ChatGPT - -# File Reading and Writing - -The File Reading and Writing page can record file read and write operations. Users can view key information such as the time of file access, the person accessing the file, and the file path, allowing them to promptly detect and handle abnormal behavior. - -![3_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650becce082cd.png) - -- You can add event information to be viewed through the `Column Options` feature of the table. For details, please refer to the section [Tracing - Call Log](../tracing/call-log/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/04-tracing/08-right-sliding-box.md b/translate/translated/06-guide/01-ee-tenant/04-tracing/08-right-sliding-box.md deleted file mode 100644 index dd0fd6bd..00000000 --- a/translate/translated/06-guide/01-ee-tenant/04-tracing/08-right-sliding-box.md +++ /dev/null @@ -1,210 +0,0 @@ ---- -title: Right Sliding Box -permalink: /guide/ee-tenant/tracing/right-sliding-box/ ---- - -> This document was translated by ChatGPT - -# Right Sliding Box - -Clicking on table rows, line chart legends, topology diagrams, and other elements on the feature page can bring up the right sliding box, which will display detailed information about the clicked data. The right sliding box offers various functionalities including knowledge graph, traffic relationships, application metrics, endpoint list, call logs, distributed tracing, network metrics, network path, flow logs, NAT tracing, events, etc., to meet different user needs. Users can select the appropriate functionality based on actual needs to view and analyze data, quickly identify and address issues, and improve work efficiency. - -Next, we will introduce each feature in detail. - -## Knowledge Graph - -The knowledge graph displays all tags associated with the clicked data object in both list and topology forms. - -![Knowledge Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab0460298a.png) - -- **Knowledge List**: Displays all tags associated with the clicked data object in key-value pairs, categorized into `Universal Tag`, `Custom Tag`, and `others`. - - Note: If the clicked data is from the `call series page`, the categories will be distinguished between Client and Server. - - Operations: Supports searching, category filtering, and empty value filtering for key-value pairs. - - Search: Supports quick search within `Select All` data. - - Left-side category filtering: Check the categories to display the corresponding content quickly. - - Show/Hide empty tags: Display or hide tags with data value `--`. - - Hover over the tag and click the `copy` icon to quickly copy. - - Paste the copied content into the search bar, which can convert it into a search tag. For details on using search tags, refer to the [Service Search Box](../query/service-search/) section. -- **Knowledge Graph**: Displays the relationships of tags in a star topology structure. Clicking on a node will display associated nodes. - -## Traffic Relationships - -The upper part of the traffic relationships displays upstream and downstream metrics of the clicked data object in a table. Clicking on a table row will use DeepFlow's self-developed flow tracing algorithm to trace the `access data` through observation points in the virtual or physical network. - -![Traffic Relationships](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fad66aac.png) - -- **① Dropdown Box:** Click the dropdown box to select the object to view traffic relationships. If the clicked data is from the `path data page`, there will be two objects. -- **② Role:** Check `as client` or `as server` to select the metrics when the current object acts as a `client` or `server`. - -![Virtual-Link Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fae1d9bb.png) - -The link topology is sorted from left to right according to the observation points the `access data` flows through, such as client application -> client process -> client network card -> ... -> server network card -> server process -> server application. - -- Note: Each `node` on the topology represents aggregated information from the same observation point. -- Hover: View metric information. - -![Virtual-Detail Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fafcb207.png) - -The detail table displays detailed information of each observation point for the `access data`, including resource belonging to the observation point, data collection location, tunnel information, and metric details. - -- Clicking on a row will display detailed information of the clicked observation point in the right sliding box. - -![Virtual-Bar Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240515664460986c6da.png) - -The bar chart displays data consistent with the link topology, providing a more intuitive comparison of data sizes by switching the topology display to a bar chart. - -- **① Latency Difference:** For example, the above chart shows the response latency metrics of access data at various observation points, where the differences between bars clearly indicate a significant latency bottleneck from the `client container node` to the `server container node` compared to other locations. - -![Physical Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fb6e1309.png) - -When `access data` flows through the `physical network`, it will display the metrics collected at each physical collection point in a physical topology form. - -- The data of `nodes` is obtained from the corresponding `network location`. If no data is collected at the corresponding `network location` for the current `access data`, it will be displayed as empty. - -## Application Metrics - -Application metrics display the aggregated values of application metrics over a period in an overview chart. It also supports adding multiple line charts to show the trend of application metrics over time. - -![Application Metrics](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab046aa9e5.png) - -- **Metric Curve** - - Metric Name: Click to select the metric line chart to display. - - Aggregation Function: Supports function calculations on selected metrics. - - Grouping: Supports grouping the current data. For example, if viewing the response latency of a service, you can further view the response latency of each application protocol by adding `l7_protocol` as a subgroup. - - Enable/Disable Tip Sync: When enabled, you can view the metrics at the same time point across all line charts. -- Click the time component in the upper right corner to filter data by time. - -## Endpoint List - -The endpoint list displays the metrics grouped by `endpoints`. - -![Endpoint List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445f9f99c90.png) - -- Clicking on a table row will take you to the `call logs` page to view the call log information of that `endpoint`. For details, refer to the [Call Logs] section. -- Click the time component in the upper right corner to filter data by time. - -## Call Logs - -Call logs display detailed call logs of the clicked data in trend analysis charts and tables. - -![Call Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab2c948e98.png) - -- Trend Analysis Chart: Displays the log collection situation within a certain time range, visualizing the data. Users can select any time period on the trend analysis chart to zoom in and view the log situation within that period. -- Log Detail Table: Displays log information in a table format, such as client, server, application protocol, request type, request domain, etc. It can dynamically display based on the trend analysis chart. For details, refer to the [Table](../dashboard/panel/table/) section. - - Clicking on a table row will take you to the detail page of that log. For details, refer to the [Call Log Details] section. -- Click the icon in the upper right corner of the trend analysis chart to open the `call logs` page in a new window for custom searches. -- Click the time component in the upper right corner to filter data by time. - -## Call Log Details - -Call log details display the response latency, application protocol, request type, request resource, and response status between the client and server of the request data at the top of the page. Below the basic information, two buttons perform different operations based on the source of the log data. Below the buttons, the corresponding tags and metrics of the log are displayed in key-value pairs for quick information retrieval. - -![Call Log Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab59c80b81.png) - -- Operation Buttons: - - View Flow Logs: Only for logs with the signal source as `Packet`, view detailed flow log information. For details, refer to the [Flow Log Details] section. - - Distributed Tracing: Supports initiating distributed tracing for logs with the signal source as `eBPF` or `OTel`. For details, refer to the [Distributed Tracing](../dashboard/panel/flame/) section. -- For tag search, filtering, and other `operations`, refer to the [Knowledge Graph] section. - -## Network Metrics - -Network metrics display the aggregated values of network metrics over a period and show the trend of network metrics over time in line charts. - -![Network Metrics](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab47494a85.png) - -- Metric Name: Click to select the metric line chart to display. -- Aggregation Function: Supports function calculations on selected metrics. -- Grouping: Supports grouping the current data. For example, if viewing the traffic size of a cloud server, you can further view the traffic size of each port of the cloud server by adding `server_port` as a subgroup. -- Enable/Disable Tip Sync: When enabled, you can view the metrics at the same time point across all line charts. -- For details on using line charts, refer to the [Line Chart](../dashboard/panel/line/) section. -- Click the time component in the upper right corner to filter data by time. - -## Flow Logs - -Flow logs record detailed information of each flow at a minute granularity and display the clicked data's flow logs in trend analysis charts and tables. - -![Flow Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cbef482a14.png) - -- Trend Analysis Chart: Displays the flow log collection situation within a certain time range, visualizing the data. Users can select any time period on the trend analysis chart to zoom in and view the log situation within that period. -- Log Detail Table: Displays flow log information in a table format, such as client, server, application protocol, request type, request domain, etc. It can dynamically display based on the trend analysis chart. For details, refer to the [Table](../dashboard/panel/table/) section. - - Left-side Quick Filter: Allows filtering flow logs by conditions through the left sidebar. For details, refer to the [Left-side Quick Filter](../query/left-quick-filter/) section. - - Clicking on a table row will take you to the detail page of that log. For details, refer to the [Flow Log Details] section. -- Click the icon in the upper right corner of the trend analysis chart to open the `flow logs` page in a new window. -- Click the time component in the upper right corner to filter data by time. - -## Flow Log Details - -Flow log details further display the information of flow logs. For logs with the protocol as `TCP`, TCP sequence tracing and NAT tracing can be performed. - -![Flow Log Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cf395cfbd9.png) - -- Operation Buttons: - - TCP Sequence Diagram: Displays detailed information of each TCP packet header, clearly showing the process of TCP connection establishment, data transmission, and connection closure. For details, refer to the [TCP Sequence Diagram Analysis] section. - - NAT Tracing: Initiates tracing using the `five-tuple`. For details, refer to the [NAT Tracing Details] section. - - PCAP Download: If there is a matching PCAP policy for the current data, it can be downloaded. - - Call Logs: Displays the current call log information. See the figure below for [14-Call Logs]. -- For `operations`, refer to the [Knowledge Graph] section. - -![Call Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cf4122132c.png) - -- Consists of a call trend analysis chart and a call log table. - - Trend Analysis Chart: Displays the number of request calls within a certain time period. - - Call Log Table: Displays detailed information of the call logs. Clicking on a table row will take you to the `call log details` page for more information. - -## TCP Sequence Diagram - -The TCP sequence diagram displays detailed information of each TCP packet header in trend analysis charts and tables, including timestamps, direction, Flag, Seq, Ack, etc., during the TCP connection establishment, data transmission, and connection closure processes, for further analysis and diagnosis of network communication issues. - -![17-TCP Sequence Diagram](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650aba4f61574.png) - -- Click the time component in the upper right corner to filter data by time. -- Sequence Table - - Time, Seq, and Ack can be toggled between relative and absolute values by clicking the buttons in the list header. - - Interval time is the time difference between the current row and the previous row. - - PCAP Download: If the current flow matches the PCAP policy, the PCAP file can be downloaded. - -## NAT Tracing - -NAT tracing can initiate tracing for any TCP four-tuple or five-tuple, using DeepFlow's self-developed algorithm to automatically trace the traffic before and after NAT. The NAT tracing page displays the metrics corresponding to the clicked data's four-tuple in a table. Clicking `trace` will initiate tracing for the four-tuple. - -![NAT Tracing](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab59bad70a.png) - -- Clicking on a table row will initiate tracing for that data. For details, refer to the [NAT Tracing Details] section. -- Click the icon in the upper right corner to open the `NAT tracing` page in a new window for custom searches. -- Click the time component in the upper right corner to filter data by time. - -## NAT Tracing Details - -NAT tracing details are divided into three main parts: the top header displays related information of the initiated tracing flow; the left side shows the traced network topology, consisting of virtual network topology and physical network topology; the right side shows the corresponding traffic topology. - -![NAT Tracing Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650aba4daabc0.png) - -- Traffic Topology: Displays all traced traffic in a `free topology` form. Clicking on a `line` will show detailed information of the corresponding path. -- Network Topology: Displays the observation points of the traced traffic in a `waterfall topology` form. - - Virtual Network Topology: Displays the observation points the traced traffic passes through in the virtual network, such as client network card, client container node, server container node, server network card. - - Physical Network Topology: Displays the network locations the traced traffic passes through in the physical network. - - Note: Physical network topology is displayed only if the traced traffic passes through the physical network. - - Operations: - - Supports hovering over `nodes` and `lines` to view more detailed information, including observation points, detailed information of network locations, tunnel information, and metrics. -- Operation Buttons: - - Modify Metrics: Supports switching metric queries. - - Show/Hide Difference: Displays or hides the main metric difference between adjacent nodes in the `network topology`. - - Show/Hide Full Name: Displays or hides the full name of `nodes` in the topology. - - Settings: - - Add to Dashboard: Supports adding the panel to the dashboard. - - View API: Basic operations of the panel, refer to the [Traffic Topology - Settings](../dashboard/panel/topology/) section. - - Close Thumbnail: Enable/disable the thumbnail in the lower left corner of the topology. - - Matching Algorithm Degree: Supports adjusting the parameters of the NAT tracing algorithm. -- Click the time component in the upper left corner to filter data by time. - -## Resource Change Events - -Displays resource change events of the clicked data object in trend analysis charts and tables. - -![Resource Change Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fa11ff96.png) - -## File Read/Write Events - -Displays file read/write events of the clicked data object in trend analysis charts and tables. - -![File Read/Write Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fa2b14d2.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/05-profiling/01-continue-profile.md b/translate/translated/06-guide/01-ee-tenant/05-profiling/01-continue-profile.md deleted file mode 100644 index fd5f6585..00000000 --- a/translate/translated/06-guide/01-ee-tenant/05-profiling/01-continue-profile.md +++ /dev/null @@ -1,40 +0,0 @@ ---- -title: Continuous Profiling -permalink: /guide/ee-tenant/profiling/continue-profile/ ---- - -> This document was translated by ChatGPT - -# Continuous Profiling - -For the functional principles, see [Core Features - Explanation of Continuous Profiling](../../../features/continuous-profiling/auto-profiling) - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146642dfb068b35.png) - -- **① Page Search**: Search bar, search snapshots, and other functions. For details, please refer to the chapter [Tracing - Resource Analysis](../tracing/service-list/) -- **② Quick Filter**: Supports filtering by `Application List` and `Profiling Type` - - Application List: Displays the reported service list information, with the default display format as **Service Name (Language Type)** - - Profiling Type: Displays the `Profiling Type` supported by the profiling method configured on the client - - Select the `Profiling Type` supported by the current application, different profiling types represent different semantics -- **③ Display Switch**: Switch the display mode of performance profiling data. Currently supports: Flame Graph, List, and Simultaneous Display - - Flame Graph: Displays the function call stack in the form of a flame graph - - **⑥ Tip**: Hover to view information about the Span in the flame graph - - Function Type: - - K: Linux kernel function - - L: Function in a dynamic link library - - A: Business function of the application - - P: Process - - T: Thread, only appears in the second layer of the flame graph - - ?: Unknown, function name translation failed. For detailed explanation, see [Core Features - Continuous Profiling - Viewing Data - About Function Type](../../../features/continuous-profiling/data/) - - Span Name - - Total Consumption: The percentage of total consumption of the Span relative to the root (the first line of the flame graph) - - Self Consumption: The percentage of self-consumption of the Span relative to the root (the first line of the flame graph) - - Operation: Click to zoom in and view the call stack of the clicked Span; click again on a blank area to return to the original state - - Table: Displays the function's `Self Consumption` and `Total Consumption` in a list format - - Default sorted by `Self Consumption` in descending order - - Simultaneous Selection: View performance profiling data in both `Flame Graph` and `Table` formats simultaneously - - Clicking on a function in the table will highlight it in the flame graph -- **④ Flame Graph Name Display**: The flame graph can choose to display the head or tail of the name -- **⑤ Data Filter**: Input characters to filter data in the table \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/01-overview.md b/translate/translated/06-guide/01-ee-tenant/06-network/01-overview.md deleted file mode 100644 index 71941eef..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/01-overview.md +++ /dev/null @@ -1,21 +0,0 @@ ---- -title: Overview -permalink: /guide/ee-tenant/network/overview/ ---- - -> This document was translated by ChatGPT - -# Overview - -The DeepFlow network module offers a wide range of features to support users in real-time monitoring of network path traffic and network performance. It includes functionalities such as resource node information, flow log data, NAT tracing, PCAP download, traffic distribution, and resource inventory. It enables real-time monitoring, analysis, and evaluation of performance parameters like data traffic, latency, and packet loss rate, allowing for the timely detection and resolution of network congestion, faults, and security issues. This improves network stability and reliability, ensuring efficient network operation. - -The following sections will provide detailed instructions and explanations for each page. - -- [Resource Analysis](./service-statistics/) -- [Path Analysis](./network-path/) -- [Topology Analysis](./network-map/) -- [Flow Log](./flow-log/) -- [NAT Tracing](./NAT-traversal/) -- [Resource Inventory](./resource-inventory/) -- [PCAP Strategy](./pacp-strategy/) -- [PCAP Download](./pcap-download/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/02-service-statistics.md b/translate/translated/06-guide/01-ee-tenant/06-network/02-service-statistics.md deleted file mode 100644 index 78ef382e..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/02-service-statistics.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Resource Analysis -permalink: /guide/ee-tenant/network/service-statistics/ ---- - -> This document was translated by ChatGPT - -# Resource Analysis - -Resource analysis displays network traffic-related information at various network locations in the form of line charts and lists. Users can quickly obtain the network traffic situation and the real-time status of each network node. - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c09d0c6aea.png) - -- For details on page button functions, please refer to the section [Tracing - Resource Analysis](../tracing/service-list/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/03-network-path.md b/translate/translated/06-guide/01-ee-tenant/06-network/03-network-path.md deleted file mode 100644 index 62248657..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/03-network-path.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Path Analysis -permalink: /guide/ee-tenant/network/network-path/ ---- - -> This document was translated by ChatGPT - -# Path Analysis - -Path analysis displays traffic information collected at network locations in the data link in the form of line charts and tables. Utilizing DeepFlow's self-developed flow tracing algorithm, it can trace network flows or application calls through observation points in virtual or physical networks. - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac4cf837e7.png) - -- For details on page button functions, please refer to the section [Tracing - Path Analysis](../tracing/service-statistics/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/04-network-map.md b/translate/translated/06-guide/01-ee-tenant/06-network/04-network-map.md deleted file mode 100644 index a42759fe..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/04-network-map.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Topology Analysis -permalink: /guide/ee-tenant/network/network-map/ ---- - -> This document was translated by ChatGPT - -# Topology Analysis - -The topology analysis page displays the relationships of observation points in virtual or physical networks through a topology format. - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac4d081034.png) - -- For details on page button functions, please refer to the section [Tracing - Topology Analysis](../tracing/path-topology/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/05-flow-log.md b/translate/translated/06-guide/01-ee-tenant/06-network/05-flow-log.md deleted file mode 100644 index 72ea8afe..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/05-flow-log.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Flow Log -permalink: /guide/ee-tenant/network/flow-log/ ---- - -> This document was translated by ChatGPT - -# Flow Log - -Flow logs record detailed information of each flow on a minute-by-minute basis, displaying flow log data through trend analysis charts and tables. The data obtained from network positions on the link is parsed layer by layer, automatically tagged through AutoTagging, and then organized to produce flow logs. - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac4d20944c.png) - -- For details on page button functions, please refer to the section [Tracing - Call Log](../tracing/call-log/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/06-NAT-traversal.md b/translate/translated/06-guide/01-ee-tenant/06-network/06-NAT-traversal.md deleted file mode 100644 index e8ce6ad8..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/06-NAT-traversal.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -title: NAT Tracing -permalink: /guide/ee-tenant/network/NAT-traversal/ ---- - -> This document was translated by ChatGPT - -# NAT Tracing - -NAT tracing can initiate tracing for any TCP quadruple or quintuple, automatically tracing pre- and post-NAT traffic using DeepFlow's proprietary algorithm. The NAT tracing page displays the metrics corresponding to the clicked quadruple in a table format. Clicking `Tracing` initiates tracing for the quadruple. - -## Overview - -![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566442f3ee84e5.png) - -- For details on page button functions, please refer to the section 【[Tracing - Call Log](../tracing/call-log/)】. -- Network tracing table: Displays client, server, group information, traffic rate, TCP retransmission ratio, TCP connection failures, and TCP connection delay in a table format. - - Operations: - - Click on a row: Initiates tracing for the data. For details, please refer to the section 【[Tracing - Right Sliding Box - NAT Tracing Details](../tracing/right-sliding-box/)】. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/07-resource-inventory.md b/translate/translated/06-guide/01-ee-tenant/06-network/07-resource-inventory.md deleted file mode 100644 index 43be5bad..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/07-resource-inventory.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Resource Inventory -permalink: /guide/ee-tenant/network/resource-inventory/ ---- - -> This document was translated by ChatGPT - -# Resource Inventory - -Resource inventory involves the auditing and management of various resources required by applications within the system, including computing resources, network resources, storage resources, etc. Typically, these resources are provided to applications in a virtualized manner. Through resource inventory, system administrators can better understand the allocation and utilization of resources, thereby more efficiently managing and optimizing resources, improving system performance and security, and also saving costs. - -## Overview Introduction - -![Overview Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac6b17b985.png) - -- For details on the functionality of page buttons, please refer to the section 【[Tracing - Call Log](../tracing/call-log/)】 \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/08-pacp-strategy.md b/translate/translated/06-guide/01-ee-tenant/06-network/08-pacp-strategy.md deleted file mode 100644 index 3352b312..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/08-pacp-strategy.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -title: PCAP Strategy -permalink: /guide/ee-tenant/network/pacp-strategy/ ---- - -> This document was translated by ChatGPT - -# PCAP Strategy - -The PCAP strategy supports setting policies or rules for capturing network data packets. The strategy allows setting the network location for packet capture, the collector, filtering rules, payload truncation, etc., specifying which types of data packets should be captured and how to filter out unnecessary data packets. - -## Overview - -![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac6b204ac3.png) - -- **① New**: Supports creating a new PCAP strategy. For details, please refer to the [New Strategy] section. -- **② Enable/Disable**: Choose to enable or disable the current PCAP strategy. Once enabled, data filtering and capturing will commence. -- **③ View Collected Traffic**: Click to jump to the PCAP download page to view the traffic data collected by this strategy. For details, please refer to the [PCAP Download](./pcap-download/) section. -- **④ Edit**: Modify the selected strategy. -- **⑤ Delete**: Delete the strategy. - -### New Strategy - -![New Strategy](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac6b31a676.png) - -- Name: Required, the name of the PCAP strategy. -- Network Location: Required, select the network location where data will be captured. -- Collector: Choose the supported collector based on the selected network location. -- Collection Point Filter: Required, can be selected from computing resources, network resources, or container resources. - - Different categories of resources support further selection of specific resource information. For example, if an IP address under network resources is selected, the IP address to be filtered must be provided. -- VPC: Optional, filter based on requirements. -- Protocol: Optional, filter based on requirements. -- Port: Optional, filter based on requirements. -- Peer: Disabled by default, supports filtering peer data. - - Peer Filter: Required, please refer to `Collector Filter` for details. - - Port: Optional, filter based on requirements. -- Payload Truncation: Enter the size of the traffic to be truncated, in bytes. - - The default is 0, meaning no truncation. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/06-network/09-pcap-download.md b/translate/translated/06-guide/01-ee-tenant/06-network/09-pcap-download.md deleted file mode 100644 index 85e2ef5c..00000000 --- a/translate/translated/06-guide/01-ee-tenant/06-network/09-pcap-download.md +++ /dev/null @@ -1,22 +0,0 @@ ---- -title: PCAP Download -permalink: /guide/ee-tenant/network/pcap-download/ ---- - -> This document was translated by ChatGPT - -# PCAP Download - -PCAP Download displays the data of enabled PCAP policies within a certain time range in the form of lists and trend analysis charts. - -## Overview - -![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac82daa46d.png) - -- **① PCAP Policy Dropdown**: The dropdown shows all PCAP policies, allowing for policy selection. - - If no selection is made, data from all PCAP policies will be displayed. -- **② Time**: Select the time range to view. If no PCAP policy is enabled during this time, no data will be available. -- For page button functions, please refer to the section 【[Tracing - Call Log](../tracing/call-log/)】. -- Log Details: Displays the traffic log information captured under the PCAP policy within a certain time range in a tabular form. - - Operations: - - Click Row: Click on a data row to view the details of that data in a right sliding box. For details on usage, please refer to the section 【[Right Sliding Box - Flow Log Details](../tracing/right-sliding-box/)】. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/07-metrics/01-overview.md b/translate/translated/06-guide/01-ee-tenant/07-metrics/01-overview.md deleted file mode 100644 index 0d78ef49..00000000 --- a/translate/translated/06-guide/01-ee-tenant/07-metrics/01-overview.md +++ /dev/null @@ -1,18 +0,0 @@ ---- -title: Overview -permalink: /guide/ee-tenant/metrics/overview/ ---- - -> This document was translated by ChatGPT - -# Overview - -DeepFlow integrates with Prometheus data to display infrastructure status, CPU, memory, and other metrics in a list format. Currently, it supports displaying data for `hosts` and `containers`. Additionally, it supports searching for metric data information and adding metric templates to specified data tables. - -- [Hosts](./host/) -- [Containers](./container/) -- [Metrics Viewing](./metrics-viewing/) -- [Metric Summary](./metric-summary/) -- [Metrics Template](./metrics-template/) - -Note: When using `hosts` or `containers`, Prometheus node_exporter data needs to be pushed to DeepFlow. For the push method, refer to [Integrating Prometheus Data](../../../integration/input/metrics/prometheus/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/07-metrics/02-host.md b/translate/translated/06-guide/01-ee-tenant/07-metrics/02-host.md deleted file mode 100644 index 180093e9..00000000 --- a/translate/translated/06-guide/01-ee-tenant/07-metrics/02-host.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Host -permalink: /guide/ee-tenant/metrics/host/ ---- - -> This document was translated by ChatGPT - -# Host - -Displays the host's name, instance IP, operating system, CPU usage, MEM usage, and system load in a list format. - -![01-Host](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310caa83d06.png) - -Clicking on a row will display historical metrics in a right-slide panel. - -![02-Right Slide Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310cab63d91.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/07-metrics/03-container.md b/translate/translated/06-guide/01-ee-tenant/07-metrics/03-container.md deleted file mode 100644 index 6f16cd6c..00000000 --- a/translate/translated/06-guide/01-ee-tenant/07-metrics/03-container.md +++ /dev/null @@ -1,18 +0,0 @@ ---- -title: Container -permalink: /guide/ee-tenant/metrics/container/ ---- - -> This document was translated by ChatGPT - -# Container - -Displays the container POD name, instance IP, associated Node, namespace, restart count statistics, status, and total runtime in a list format. - -![01-容器](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310caccb45f.png) - -Clicking on a row will display historical metrics in a right slide-out panel. - -![02-右滑框](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310cac1f0c1.png) - -**① Switch container dropdown**: Use this dropdown to quickly switch the container of the selected POD. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/07-metrics/05-metric-summary.md b/translate/translated/06-guide/01-ee-tenant/07-metrics/05-metric-summary.md deleted file mode 100644 index a5c2345f..00000000 --- a/translate/translated/06-guide/01-ee-tenant/07-metrics/05-metric-summary.md +++ /dev/null @@ -1,15 +0,0 @@ ---- -title: Metric Summary -permalink: /guide/ee-tenant/metrics/metric-summary/ ---- - -> This document was translated by ChatGPT - -# Metric Summary - -Switch between `Metric Sets` to view the Metrics and Tag related data stored in the corresponding data tables. - -![2_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650bb8734ad16.png) - -- **① Metric Sets**: Supports switching to find data tables -- **② Tab Switching**: Switch tabs to view the Metrics and Tags stored in the corresponding data tables \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/07-metrics/06-metrics-template.md b/translate/translated/06-guide/01-ee-tenant/07-metrics/06-metrics-template.md deleted file mode 100644 index be03b0ea..00000000 --- a/translate/translated/06-guide/01-ee-tenant/07-metrics/06-metrics-template.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -title: Metrics Template -permalink: /guide/ee-tenant/metrics/metrics-template/ ---- - -> This document was translated by ChatGPT - -# Metrics Template - -A series of metric collections established for different data tables, which can be used in the chart editing query module. - -## Overview - -Displays the metric templates that the current account can view in a list format, and supports operations such as editing and deleting. Some data tables include default templates, which are created by the system and cannot be edited or deleted. - -![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240514664334ae9febc.png) - -- **① Create Template**: Create a metric template. For usage details, please refer to the [Create Metric Template](#create-metric-template) section. -- **② Switch Data Table**: Switch data tables to view the metric templates under the corresponding data table. -- **③ Delete**: Delete a metric template. -- **④ Edit**: Edit a metric template. -- **⑤ Row Expand**: Click on the table row to quickly expand and view the metrics included in the template. For details, please refer to the image below [01-Row Expand](#01-row-expand). - -![01-Row Expand](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643341c20532.png) - -## Create Metric Template - -![02-Create Metric Template](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466433419ccda4.png) - -- Template Name: Required, the name of the newly created template. -- Team: Required, select the team that can view the template. -- Data Table: Required, select the data table to which the template will be added. -- Metrics: Select the metrics to be added. - - Supports setting aggregation operators, metric aliases, units, thresholds, etc. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/08-log/01-log.md b/translate/translated/06-guide/01-ee-tenant/08-log/01-log.md deleted file mode 100644 index cc2eed94..00000000 --- a/translate/translated/06-guide/01-ee-tenant/08-log/01-log.md +++ /dev/null @@ -1,26 +0,0 @@ ---- -title: Logs -permalink: /guide/ee-tenant/log/log/ ---- - -> This document was translated by ChatGPT - -# Logs - -Logs display information related to system and application processes. DeepFlow enhances log functionality through advanced log processing technologies and methods, improving the completeness and accuracy of data tracing collection, and strengthening storage and display capabilities to meet enterprise-level application needs. - -![01-日志](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024061966729f8a3a45b.png) - -- **① Time Selector**: Supports time filter queries. For details, please refer to the section [Dashboard Details - Time Selector](../dashboard/use/) -- **② Search Snapshot**: Supports saving search conditions as snapshots. For details, please refer to the section [Query - Search Snapshot](../query/history/) -- **③ Search Save and Settings**: - - Search Save: Supports quickly saving the search conditions of the current page. For details, please refer to the section [Query - Search Snapshot](../query/history/) - - Settings: A collection of page setting operations - - Database Fields: Supports viewing tags and metrics in the data table used on the current page - - Name Default/Full Display: Full or default display of legend names -- **④ Service Search Box**: Supports searching or grouping by Tag. For details, please refer to the section [Query](../query/overview/) -- **⑤ Left Quick Filter**: Supports quick filtering of `Application Services` and `Log Levels`. For details, please refer to the section [Query - Left Quick Filter](../query/left-quick-filter/) -- **⑥ Area Query**: Supports quickly switching query area data -- **Table Operations**: - - Click Row: Clicking on the table data allows you to quickly enter the right sliding box to view related information of the corresponding application service. For details, please refer to the section [Tracing - Right Sliding Box](../tracing/right-sliding-box/) - - ⑦ Copy: Table content copy function. After clicking, you can select the content format to copy. For configuration details, please refer to the section [Table](../dashboard/panel/table/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/09-alert/01-overview.md b/translate/translated/06-guide/01-ee-tenant/09-alert/01-overview.md deleted file mode 100644 index 7c5cb734..00000000 --- a/translate/translated/06-guide/01-ee-tenant/09-alert/01-overview.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Overview -permalink: /guide/ee-tenant/alert/overview/ ---- - -> This document was translated by ChatGPT - -# Overview - -Supports user-defined alert policies, allowing the monitored data to be pushed to the metric object in real-time. Additionally, related information can be viewed on the alert event page. - -DeepFlow's alerts are divided into three main pages, which will be detailed in the following sections. - -- [Alert Policy](./alert-policy/) -- [Push Endpoint](./push-endpoint/) -- [Alert Event](./alert-event/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/09-alert/02-alert-policy.md b/translate/translated/06-guide/01-ee-tenant/09-alert/02-alert-policy.md deleted file mode 100644 index 912a31f3..00000000 --- a/translate/translated/06-guide/01-ee-tenant/09-alert/02-alert-policy.md +++ /dev/null @@ -1,45 +0,0 @@ ---- -title: Alert Policy -permalink: /guide/ee-tenant/alert/alert-policy/ ---- - -> This document was translated by ChatGPT - -# Alert Policy - -The formulation of an alert policy is used to identify and respond to abnormal conditions in programs or services to ensure the normal operation of the system and the stability of business processes. -The alert policy page displays information about all alert policies in a list format. - -![00-Overview.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566447b93e1025.png) - -- Number of Alerts: Displays the number of alert data generated after the corresponding alert policy takes effect. You can click to jump to the [Alert Event](./alert-event/) page for further information. -- Status: You can choose to enable or disable the policy. -- Actions: - - Edit: Edit the corresponding alert policy, supporting modifications to the alert level and push endpoints. For details, please refer to the [Edit Alert Policy] section. - - Delete: Only supports deleting alert policies that are `disabled`. - -## Edit Alert Policy - -The alert policy requires filling in the configuration information for three modules: basic information, monitoring configuration, and notification configuration to generate the required alert policy. - -![01-Edit Alert Policy.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566447b8697e3d.png) - -- Basic Information - - Alert Name: Required, fill in the corresponding alert name. - - Team: Required, select the team organization that can view the policy. - - Add Tags: Support adding tags to the alert policy. - - Level: The importance level of the alert policy. - -![02-Edit Alert Policy.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566447b880ecc2.png) - -- Monitoring Configuration - - Monitoring Frequency: The time interval between two data monitoring sessions. - - Monitoring Interval: The time range for data queries each time the policy is executed. - - Options include `1 minute`, `5 minutes`, `15 minutes`, `30 minutes`, `1 hour`. - - Monitoring Metrics: Select the data metrics to be monitored. - - Event Level: Based on the set conditions, monitoring events can be classified into six levels: `Critical`, `Error`, `Warning`, `Recovery`, `Info`, `No Data`. - - Recovery: When the results of consecutive X monitoring events do not meet any of the conditions for `Critical`, `Error`, `Warning`, or `No Data`, a `Recovery Event` is generated. - - Info: When enabled, if the results of the monitoring event do not meet any of the conditions for `Critical`, `Error`, `Warning`, `Recovery`, or `No Data`, an `Info Event` is generated. - - No Data: When enabled, if the monitoring event has no data, a `No Data Event` is generated. - - Notification Configuration - - Push Endpoints: Select the objects to push notifications to. Multiple objects can be selected. For configuration details, please refer to the [Push Endpoints](./push-endpoint/) section. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/09-alert/03-push-endpoint.md b/translate/translated/06-guide/01-ee-tenant/09-alert/03-push-endpoint.md deleted file mode 100644 index 3f111b05..00000000 --- a/translate/translated/06-guide/01-ee-tenant/09-alert/03-push-endpoint.md +++ /dev/null @@ -1,100 +0,0 @@ ---- -title: Push Endpoints -permalink: /guide/ee-tenant/alert/push-endpoint/ ---- - -> This document was translated by ChatGPT - -# Push Endpoints - -Push endpoints are used to receive and process alerts in systems or services. Currently, four push methods are supported: Email push, HTTP push, Kafka push, PCAP strategy, and Syslog push. - -Next, we will introduce these five push methods separately. - -## Email Push - -Send alert events to a specified email address, allowing you to stay informed about alerts by checking your email. - -![Email Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230428644b76b451e05.png) - -- Create a new Email Push: Fill in the relevant information to successfully create it, which can be used when [creating an alert policy](./alert-policy/) -- List - - Associated Alert Policies: Click the number to jump to the [Alert Policies](./alert-policy/) page to view the alert policies using this push endpoint - - Edit: Supports editing the push endpoint - - Delete: Supports deleting the push endpoint - -### Create a New Email Push - -![Create a New Email Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b62681d2f.png) - -- Email: Fill in the email address to push to -- Push Title: Optional, supports filling in the email title -- For other fields, please refer to the [Create a New Kafka Push](#create-a-new-kafka-push) section - -## HTTP Push - -HTTP push sends data to a specified URL address via the HTTP protocol. - -![HTTP Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230428644b7a5c0c7bd.png) - -- For page button usage, please refer to the [Email Push](#email-push) section - -### Create a New HTTP Push - -![Create a New HTTP Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b6260915a.png) - -- Push Method: Required, supports `POST`, `PUT`, and `PATCH` methods, default is `POST` -- Push URL: Required, protocol name is case-insensitive, supports HTTP and HTTPS protocols, `HTTPS` can be unidirectional authentication - - Note: Supports Jinja template rendering, e.g., `http://10.0.0.1/{{policy_level}}` -- Header: Enter `HTTP` key-value pairs -- For other fields, please refer to the [Create a New Kafka Push](#create-a-new-kafka-push) section - -## Kafka Push - -Kafka push supports pushing alert events to Kafka. - -- For page button usage, please refer to the [Email Push](#email-push) section - -### Create a New Kafka Push - -![Create a New Kafka Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051666456bfb6ebdc.png) - -- Name: Required, fill in the name of the push endpoint -- Team: Required, select the team that can use this push endpoint -- Broker Address Pool: Required, input format `[address]:[port]`, supports multiple entries separated by commas -- Topic: Required, Kafka push topic, supports 1-256 printable characters -- SASL: Optional, authentication method, if Plain is selected, username and password need to be filled in -- Push Content: Supports Jinja template rendering, for default push content, please refer to the parameter description -- Configuration Level: Select the alert event level to receive, for alert event level description, please refer to the [Edit Alert Policy](./alert-policy/) section - - By default, all alert events except `info` will be pushed -- Push Cycle: Required, within the push cycle, alert events generated by the same monitoring object under the same alert policy will only be pushed once -- Push Frequency: The maximum number of times alert events generated by the same monitoring object under the same alert policy can be pushed. Exceeding this limit will stop further pushes - -## PCAP Strategy - -Supports adding alert policies to the PCAP strategy for alert monitoring through PCAP. - -![Create a New PCAP Strategy](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664573d598713.png) - -- For page button usage, please refer to the [Email Push](#email-push) section -- Create a New PCAP Strategy - - Associate PCAP Strategy: Required, associate the PCAP strategy with alert events. Alerts generated can be downloaded in the associated PCAP strategy -- Enable PCAP Strategy: Select the alert event level to push. If an alert event of this level is generated, it will be pushed to the associated PCAP strategy, and the PCAP strategy will be automatically enabled - - Note: By default, `fatal`, `error`, and `warning` alert events will be pushed -- Disable PCAP Strategy: Select the alert event level to push. If an alert event of this level is generated, the associated PCAP strategy will be automatically disabled - - Note: By default, `recovery` alert events will be pushed -- For other fields, please refer to the [Create a New Kafka Push](#create-a-new-kafka-push) section - -## Syslog Push - -Alert Syslog push sends alert information to the log server via the Syslog protocol. It can promptly notify operations personnel of potential system failures or security events, helping them take appropriate measures in a timely manner. - -- For page button usage, please refer to the [Email Push](#email-push) section - -### Create a New Syslog Push - -![Create a New Syslog Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b6249798c.png) - -- Push Destination: Required, input format `[forwarding protocol]://[log server address]:[port]` - - Note: Optional protocols are `UDP` and `TCP`, default is `UDP`; optional ports are `1-65535`, default port is `514` -- For other fields, please refer to the [Create a New Kafka Push](#create-a-new-kafka-push) section \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/09-alert/04-alert-event.md b/translate/translated/06-guide/01-ee-tenant/09-alert/04-alert-event.md deleted file mode 100644 index a67c46d1..00000000 --- a/translate/translated/06-guide/01-ee-tenant/09-alert/04-alert-event.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -title: Alert Events -permalink: /guide/ee-tenant/alert/alert-event/ ---- - -> This document was translated by ChatGPT - -# Alert Events - -The Alert Events page displays events generated by alert policies monitoring. For example, when the system detects security vulnerabilities, abnormal behavior, or other situations requiring user attention, corresponding alert events are generated, providing detailed descriptions and recommendations. - -Users can filter and query based on multiple dimensions such as policy level, event level, monitored object, creator, monitored object, region, event UID, etc. - -![4_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650bf3357d79c.png) - -- You can add the event information you want to view through the table's `Column Options` feature. For usage details, please refer to the section [Tracing - Call Log](../tracing/call-log/). -- Double-click a table row to display detailed information about the corresponding alert event in a pop-up window. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/10-report/01-report.md b/translate/translated/06-guide/01-ee-tenant/10-report/01-report.md deleted file mode 100644 index 3d255292..00000000 --- a/translate/translated/06-guide/01-ee-tenant/10-report/01-report.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -title: Report -permalink: /guide/ee-tenant/report/report/ ---- - -> This document was translated by ChatGPT - -# Report - -Reports can be created in the Dashboard and scheduled to be pushed to users in an offline downloadable HTML format, recording the results of the Dashboard during the report period. - -## Report Policy - -The Report Policy page displays all report policies in a list format. - -![01-report-forms.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540cc8a9690e.png) - -- **① Policy Name (Number of Reports)**: Click the policy name to navigate to the `Report Download` page to view all reports generated by this policy. -- **② Object**: Displays the name of the Dashboard that generated the current report policy. Click to navigate to the corresponding Dashboard. For details on using the Dashboard, please refer to the chapter [Dashboard - Dashboard Details](../dashboard/use/). -- **③ Enable/Disable**: Start or stop the report generation for the current policy. -- **④ Edit**: Allows modification of the current policy's name and push email. -- **⑤ Delete**: Delete the current policy. - -### Create a New Report Policy - -Next, we will detail how to create a report policy through the Dashboard. - -- First, enter the Dashboard for which you want to generate a report, click `Settings`, and select `Create New Report Policy` to create a report. - -![02-dashboard.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540cc961795d.png) - -- Second, set the report policy according to the push requirements. - - Fill in the information for the report policy name, period, report format, statistical granularity, push email, etc. - - Note: The reports generated by the policy will be pushed to the email daily. - - For example, if the period is set to a weekly report, the report for the previous week (e.g., last Tuesday to this Tuesday) will be pushed daily. The report covers from 00:00 on the start date to 24:00 on the end date. Any modifications to the report policy will take effect from the last modification before the report is generated and will affect subsequent reports from that day onwards. Previous reports will not be affected. The report email will include the report attachment, which can be downloaded by clicking. - -![03-add-report-forms.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540cca8c9511.png) - -## Report Download - -The Report Download page displays all generated reports in reverse chronological order (most recent first). - -![04-download-report-forms.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540ccc1e1fec.png) - -- **① Dashboard Name**: Displays the name of the Dashboard that generated the current report policy. Click to navigate to the corresponding Dashboard. For details on using the Dashboard, please refer to the chapter [Dashboard - Dashboard Details](../dashboard/use/). -- **② Download**: Supports downloading the CSV data of the current report. -- **③ Delete**: Delete the current report. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/02-summary.md b/translate/translated/06-guide/01-ee-tenant/11-resources/02-summary.md deleted file mode 100644 index a981c756..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/02-summary.md +++ /dev/null @@ -1,15 +0,0 @@ ---- -title: Summary -permalink: /guide/ee-tenant/resources/summary/ ---- - -> This document was translated by ChatGPT - -# Summary - -On the summary page, you can view the distribution of network and computing resources. Information about each resource is displayed in pie charts. - -![Summary](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023042464463e64633af.png) - -- Network Resources: Statistics and display of the distribution of VPCs, subnets, and external IPs on the cloud platform. -- Computing Resources: Display of the distribution of cloud platforms where cloud servers are located, their running status, and the distribution status of corresponding collectors. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/03-resource-changes.md b/translate/translated/06-guide/01-ee-tenant/11-resources/03-resource-changes.md deleted file mode 100644 index 255d8b5d..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/03-resource-changes.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -title: Change Events -permalink: /guide/ee-tenant/resources/resource-changes/ ---- - -> This document was translated by ChatGPT - -# Change Events - -On the Change Events page, users can view real-time events occurring in the system, such as file creation, file deletion, etc. Users can filter event types as needed and quickly locate events of interest. - -Users can view events that occurred within a specific time range or filter by user or resource to better understand the details and context of the events. - -![1_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650bbbf54f94c.png) - -- For details on the functionality of the page buttons, please refer to the section [Tracing - Call Log](../tracing/call-log/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/04-resource-pool.md b/translate/translated/06-guide/01-ee-tenant/11-resources/04-resource-pool.md deleted file mode 100644 index c9396019..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/04-resource-pool.md +++ /dev/null @@ -1,54 +0,0 @@ ---- -title: Resource Pool -permalink: /guide/ee-tenant/resources/resource-pool/ ---- - -> This document was translated by ChatGPT - -# Resource Pool - -The resource pool in a cloud computing environment organizes and displays resource collection information at three levels: cloud platform, region, and availability zone. By organizing resource pools in the cloud environment, the cloud computing platform can better manage its internal resources, improve resource utilization and efficiency, and meet users' personalized needs. - -Next, we will introduce the three organizations separately. - -## Cloud Platform - -The cloud platform displays the resource pools and related service information data within the cloud computing environment's cloud platform. It supports creating new cloud platforms, configuring synchronization intervals, modifications, and other functions. The page displays the data information of each added cloud platform in a table format, such as type, number of regions, number of availability zones, affiliated container clusters, resource synchronization controllers, status, and other information. - -![01-云平台](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230424644643e1209c0.png) - -- **① Action Buttons**: - - Create Cloud Platform: Supports creating new cloud platforms that have been connected - - Export CSV: Supports downloading data - - Configure Sync Interval: Sets the cloud platform synchronization rate - - Sync Time Statistics: Displays the synchronization time data of all added cloud platforms in a line chart -- **② Expand Row**: Click the data row header to further display detailed information of the corresponding cloud platform -- **③ Number of Regions**: Click to jump to the page to view the corresponding region information, please refer to the [Region] section -- **④ Number of Availability Zones**: Click to jump to the page to view the corresponding availability zone information, please refer to the [Availability Zone] section -- **⑤ Enable/Disable**: Enable or stop synchronizing data for the cloud platform -- **⑥ Edit**: Edit the data in this row -- **⑦ Delete**: Delete the data in this row - -## Region - -The region displays the geographical location of data centers within the cloud platform. It allows viewing related information of the region, such as the number of availability zones, VPCs, subnets, cloud servers, and PODs. It also supports creating new regions, modifications, and exporting data. - -![02-区域](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230424644650cec4b7f.png) - -- **① Action Buttons**: - - Create Region: Supports creating new regions - - Export CSV: Supports downloading data -- **② Number of Availability Zones**: Click to jump to the page to view the information of availability zones within the region, please refer to the [Availability Zone] section -- **③ Number of VPCs**: Click to jump to the page to view the information of VPCs within the region, please refer to the [Network Resources - VPC](./network-resources/) section -- **④ Number of Subnets**: Click to jump to the page to view the information of subnets within the region, please refer to the [Network Resources - Subnet](./network-resources/) section -- **⑤ Number of Cloud Servers**: Click to jump to the page to view the information of cloud servers within the region, please refer to the [Compute Resources - Cloud Server](./network-resources/) section -- **⑥ Number of PODs**: Click to jump to the page to view the information of container PODs within the region, please refer to the [Container Resources - Container POD](./network-resources/) section -- **⑦ Edit**: Edit the data in this row - -## Availability Zone - -The availability zone displays the data center information within the cloud platform, the region where the availability zone is located, the cloud platform, the number of cloud servers, and the number of PODs. It supports creating new availability zones, modifications, and exporting data. - -![03-可用区](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230425644783b74e992.png) - -- For page button usage, please refer to the [Region] section diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/05-computing-resources.md b/translate/translated/06-guide/01-ee-tenant/11-resources/05-computing-resources.md deleted file mode 100644 index 89416a92..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/05-computing-resources.md +++ /dev/null @@ -1,30 +0,0 @@ ---- -title: Computing Resources -permalink: /guide/ee-tenant/resources/computing-resources/ ---- - -> This document was translated by ChatGPT - -# Computing Resources - -The Computing Resources page allows you to view physical or virtual devices used for processing and executing computing tasks. You can separately view information on cloud servers and host machines. - -## Cloud Servers - -Cloud servers support users in deploying and running various applications. The Cloud Servers page provides relevant information. - -![Cloud Servers](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304256447a5dd95922.png) - -- **① Action Buttons**: - - Create Cloud Server: Supports creating a new cloud server - - Export CSV: Supports downloading data - - All Network Cards: Displays all network card-related information in a list, including IP, MAC, name, subnet, associated cloud server, etc., and supports exporting CSV -- **② Refresh Cache**: Refreshes the cache for quick data access -- **③ VPC**: Click to navigate to the page to view the VPC information of the corresponding cloud server. Please refer to the section [Network Resources-VPC](./network-resources/) -- **④ Collector Status**: Supports viewing collector-related information of the cloud server, such as basic information, configuration information, monitoring information, running logs, etc. -- **⑤ Edit**: Only supports name modification -- **⑥ Network Card List**: Supports viewing network card information under the current cloud server - -## Host Machines - -Host machines are physical servers running virtualization software on the operating system. The Host Machines page provides relevant information such as region, availability zone, related IP information, CPU, memory, network cards, collectors, etc. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/06-network-resources.md b/translate/translated/06-guide/01-ee-tenant/11-resources/06-network-resources.md deleted file mode 100644 index 57938be4..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/06-network-resources.md +++ /dev/null @@ -1,49 +0,0 @@ ---- -title: Network Resources -permalink: /guide/ee-tenant/resources/network-resources/ ---- - -> This document was translated by ChatGPT - -# Network Resources - -The Network Resources page allows you to view information related to resources in the current network architecture, including VPCs, subnets, routers, DHCP gateways, and IP addresses. Each network resource has different functions and roles and can be used to build a complete network architecture. - -## VPC - -A VPC is a virtual network environment. On the VPC page, you can view relevant data information in the virtual environment. - -![01-VPC](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448916ab2058.png) - -- For the usage of page buttons, please refer to the section [Resource Pool - Region](./network-resources/). - -## Subnet - -A subnet is a logical partition within a VPC that corresponds to an actual network address segment. Instances within a subnet can connect to other subnets or the public network through the VPC's router. - -![02-子网](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448943337299.png) - -- For the usage of page buttons, please refer to the section [Resource Pool - Region](./network-resources/). - -## Router - -A router is responsible for transmitting requests from one network to another. On the Router page, you can view information about all routers. - -![03-路由器](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023042664489e7cde5f0.png) - -- All Routing Rules: Supports viewing all routing tables, which contain tables with paths to specific addresses, including destination IP, next hop type, next hop, and the router it belongs to. -- Operations - - View Routing Rules: You can view the routing table of the current router. -- For the usage of other page buttons, please refer to the section [Resource Pool - Region](./network-resources/). - -## DHCP Gateway - -The DHCP gateway supports automatic IP address allocation. On the DHCP Gateway page, you can view information about all DHCP gateways, such as region, VPC, IP, cloud platform, deletion time, etc. - -## IP Address - -An IP address is a numerical address used to identify devices in a network. - -![04-IP地址](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448be4281264.png) - -- For the usage of other page buttons, please refer to the section [Resource Pool - Region](./network-resources/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/07-network-services.md b/translate/translated/06-guide/01-ee-tenant/11-resources/07-network-services.md deleted file mode 100644 index fb269de0..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/07-network-services.md +++ /dev/null @@ -1,45 +0,0 @@ ---- -title: Network Services -permalink: /guide/ee-tenant/resources/network-services/ ---- - -> This document was translated by ChatGPT - -# Network Services - -Network services display relevant data information of different components in the cloud computing network architecture, including security groups, NAT gateways, load balancers, peering connections, and cloud enterprise networks. - -## Security Groups - -Security groups are a virtual network security mechanism in cloud computing that controls the inbound and outbound traffic of cloud instances, thereby protecting the security of cloud instances and cloud networks. - -![01-Security Groups](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448d05507924.png) - -- All Security Group Rules: Supports viewing all security group rules, including direction, IP type, protocol type, local port range, remote port range, etc., and supports exporting CSV. -- Operations - - View Security Group Rules: Supports viewing all rules of the clicked security group. -- For the usage of other buttons on the page, please refer to the section [Resource Pool - Region](./network-resources/). - -## NAT Gateway - -NAT gateways can achieve network address mapping between different subnets and enable communication between private network instances and the internet. The NAT gateway page supports viewing and downloading related information, such as external IP, region, VPC, number of gateway rules, cloud platform, etc., and also supports viewing and downloading gateway rule information. - -- For the usage of buttons on the page, please refer to the section [Resource Pool - Region](./network-resources/). - -## Load Balancer - -Load balancers in cloud computing distribute traffic to multiple virtual machines by adjusting the traffic of virtual machines, thereby achieving load balancing and improving the availability and scalability of applications. The load balancer page supports viewing and downloading related information, such as IP, region, VPC, number of load balancers, cloud platform, etc., and also supports viewing and downloading rule information. - -- For the usage of buttons on the page, please refer to the section [Resource Pool - Region](./network-resources/). - -## Peering Connection - -Peering connections are local network connection services that support the establishment of virtual private network interconnections between geographically different VPCs. The peering connection page allows viewing related information, such as local region, local VPC, remote region, remote VPC, cloud platform, etc., and supports creating new peering connections and exporting CSV. - -- For the usage of buttons on the page, please refer to the section [Resource Pool - Region](./network-resources/). - -## Cloud Enterprise Network - -Cloud enterprise networks are a cloud computing solution that provides interconnection between private and public networks, and also enables network interconnection between multiple different cloud vendors, improving the security, stability, and reliability of data communication. The cloud enterprise network page supports viewing related information, such as associated instances, cloud platform, etc., and also supports exporting CSV. - -- For the usage of buttons on the page, please refer to the section [Resource Pool - Region](./network-resources/). \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/08-storage-services.md b/translate/translated/06-guide/01-ee-tenant/11-resources/08-storage-services.md deleted file mode 100644 index 4036f295..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/08-storage-services.md +++ /dev/null @@ -1,18 +0,0 @@ ---- -title: Storage Services -permalink: /guide/ee-tenant/resources/storage-services/ ---- - -> This document was translated by ChatGPT - -# Storage Services - -Storage services are a fundamental offering of cloud computing platforms, helping users store and manage data and other objects while providing high availability, reliability, and elastic scalability. Within storage services, you can separately view information on Cloud Database RDS and Cloud Database Redis. - -## Cloud Database RDS - -Cloud Database RDS is a high-availability, high-performance, and highly secure managed database service that offers users convenient and flexible storage services and backup recovery functions. It meets enterprise needs for data storage and management, taking on corresponding management tasks and improving enterprise efficiency. On the Cloud Database RDS page, you can view and download related information such as region, availability zone, VPC, subnet, internal IP, external IP, status, database platform, cloud platform, and more. - -## Cloud Database Redis - -Cloud Database Redis is a high-speed caching database service that supports multiple data structures, significantly enhancing data read/write speed and performance, thereby improving system performance and response speed. It can be used to meet various data caching needs and other functionalities in different scenarios. On the Cloud Database Redis page, you can view and download related information, with the list content being consistent with the [Cloud Database RDS] section. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/09-container-resources.md b/translate/translated/06-guide/01-ee-tenant/11-resources/09-container-resources.md deleted file mode 100644 index c717ae54..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/09-container-resources.md +++ /dev/null @@ -1,56 +0,0 @@ ---- -title: Container Resources -permalink: /guide/ee-tenant/resources/container-resources/ ---- - -> This document was translated by ChatGPT - -# Container Resources - -Container resources encompass various dimensions of resources. Through proper configuration and limitations, resources can be allocated and managed efficiently, ensuring that applications run stably and reliably in a container environment. In container resources, you can view information on container clusters, namespaces, container nodes, Ingress, container services, workloads, ReplicaSets, and container PODs. - -## Container Clusters - -Container clusters are a group of connected containers that achieve high availability and scalability by sharing network and storage resources. On the container resources page, you can view and download related information such as management platform, availability zone, cloud platform, VPC, number of nodes, etc. - -## Namespaces - -Namespaces are used to provide isolated runtime environments for containers to avoid conflicts between them. On the namespaces page, you can view and download related information such as availability zone, number of PODs, number of ReplicaSets, clusters, etc. - -## Container Nodes - -Container nodes refer to the physical or virtual servers running container instances in a container orchestration system. On the container nodes page, you can view and download related information such as availability zone, cloud server, type, status, IP, internal routing IP, CPU, memory, cluster, number of PODs, collectors, etc. - -## Ingress - -Ingress is a way to expose HTTP and HTTPS services in a K8s cluster by creating an API object in the cluster that routes public traffic to the corresponding services. - -![01-Ingress](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448dfd092f9a.png) - -- All forwarding rules: Supports viewing and downloading all forwarding rules. The list includes protocol type, domain name, path, service, service PORT, and associated Ingress. -- Operations: - - View forwarding rules: You can view and download the forwarding rules under the current Ingress. -- For other button usage on the page, please refer to the section [Resource Pool - Region](./network-resources/). - -## Container Services - -Container services are cloud-based services that package applications and services into one or more containers and deploy them to the cloud, allowing them to run anywhere, thereby simplifying the deployment and management process of applications. - -![02-容器服务](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448e6b382c9a.png) - -- All port mappings: Supports viewing and downloading all internal application port mappings of containers to the host (node) ports, showing protocol type, node PORT, service PORT, container PORT, and associated service, and supports downloading. -- Operations: - - View port mappings: You can view and download the port mapping list of the current container. -- For other button usage on the page, please refer to the section [Resource Pool - Region](./network-resources/). - -## Workloads - -Workloads refer to applications and services hosted on the cloud platform, including multiple components, services, processes, and containers. They can run based on container or virtualization technology to ensure availability and scalability. On the workloads page, you can view and download related information such as availability zone, type, number of running PODs, desired number of PODs, number of ReplicaSets, namespace, cluster, K8s.label, etc. - -## ReplicaSet - -ReplicaSet is a controller in a K8s cluster that provides high availability, elasticity, and automated management of replicas. It ensures the availability of Pods and automatically adjusts the number of PODs when needed. On the ReplicaSet page, you can view and download related information such as availability zone, number of running PODs, desired number of PODs, workload, namespace, cluster, K8s.label, etc. - -## Container PODs - -Containers are a way to package applications and dependencies, and PODs are collections of multiple containers that can share resources and work together. On the container POD page, you can view and download related information such as service, MAC, IP, status, namespace, cluster, etc. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/11-other-resources.md b/translate/translated/06-guide/01-ee-tenant/11-resources/11-other-resources.md deleted file mode 100644 index 4fd93a54..00000000 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/11-other-resources.md +++ /dev/null @@ -1,33 +0,0 @@ ---- -title: Other Resources -permalink: /guide/ee-tenant/resources/other-resources/ ---- - -> This document was translated by ChatGPT - -# Other Resources - -Other resources include information related to other resources in the network or system, such as physical network elements, network locations, physical links, and legends. - -## Physical Network Elements - -Physical network elements refer to hardware devices in the network, including routers, switches, firewalls, gateways, etc. These devices enable data communication and transmission by connecting and exchanging network traffic. On the physical network elements page, you can view and download related information, such as region, type, etc. It also supports creating, modifying, and deleting physical network elements. - -## Network Locations - -Network locations refer to the sources or destinations of information collected during the information collection and analysis process. On the network locations page, you can view and download related data, such as region, data tags, type, VLAN tags, source IP, interface name, sampling rate, etc. It also supports creating, modifying, and deleting network locations. - -## Physical Links - -Physical links refer to the physical lines connecting network devices. Physical links have a significant impact on the stability, speed, and reliability of network transmission. - -![01-物理链路](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266449023482d95.png) - -- View Topology: Displays the logical dependency relationships of all physical links in the current list in the form of a topology diagram. -- For page button usage, please refer to the section [Resource Pool - Region](./network-resources/). - -## Legends - -Legends represent related resources in the form of icons. All legends are displayed in a list format, supporting the creation and modification of legends. - -![02-图例](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266449034569faf.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/12-system/01-overview.md b/translate/translated/06-guide/01-ee-tenant/12-system/01-overview.md deleted file mode 100644 index 7eccdf1f..00000000 --- a/translate/translated/06-guide/01-ee-tenant/12-system/01-overview.md +++ /dev/null @@ -1,15 +0,0 @@ ---- -title: Overview -permalink: /guide/ee-tenant/system/overview/ ---- - -> This document was translated by ChatGPT - -# Overview - -The system module allows users to view information about collectors, data nodes, accounts, and operation logs. - -- [Collectors](./agent/) -- [Data Nodes](./data-node/) -- [Account Management](./account-management/) -- [Operation Logs](./operation-log/) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/12-system/02-agent.md b/translate/translated/06-guide/01-ee-tenant/12-system/02-agent.md deleted file mode 100644 index 2a804b1d..00000000 --- a/translate/translated/06-guide/01-ee-tenant/12-system/02-agent.md +++ /dev/null @@ -1,89 +0,0 @@ ---- -title: Collector -permalink: /guide/ee-tenant/system/agent/ ---- - -> This document was translated by ChatGPT - -# Collector - -The DeepFlow Collector is a tool used for collecting network and application performance data, supporting the parsing of various TraceID and SpanID specifications in protocols such as HTTP and Dubbo. - -Next, we will introduce the collector module. - -## List - -The collector list displays the installation, deployment, and running status of collectors, and also supports batch operations on collectors. - -![01-列表](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4a708edd.png) - -- First row operation buttons: - - Enable: Select multiple collectors to perform batch enable operations - - Disable: Select multiple collectors to perform batch disable operations - - Register: Select multiple collectors to perform batch registration operations - - Add to Collector Group: Select multiple collectors to perform batch add to collector group operations - - Export CSV: Select multiple collectors to perform batch export CSV operations -- Collector List - - Name: Click to jump to the collector details page to view collector information - - Basic Information: Displays the current basic information of the collector, the environment configuration of the collector, status information, and information about the running environment of the collector - - Configuration Information: Displays detailed information about the collector group configuration corresponding to the collector - - Monitoring Data: Displays all monitoring charts on the current collector details page - - Running Logs: Supports viewing the running logs of non-primary region collectors in multi-region deployment scenarios. Displays all logs recorded with the collector in ES, with WARN logs marked in yellow and ERR logs marked in red - - Group: The group to which the collector belongs. Click to jump to the [Group] page. Collectors not assigned to a group belong to the default group. Please refer to the [Group] section for details - - Type: Displays the current running environment of the collector - - KVM: The collector Trident process runs on the host machine (e.g., KVM) - - Container-V/Container-P: The collector Trident runs as a DaemonSet on each container node (K8S Node) - - ESXi: The collector Trident process runs in a dedicated virtual machine on vSphere ESXi, collecting mirror traffic from all business virtual machines on ESXi - - Dedicated Server: The collector Trident process runs on a dedicated server, collecting mirror traffic from physical switches - - Workload-V: The collector Trident process runs inside the business virtual machine - - Workload-P: The collector Trident process runs inside the business bare metal server - - Tunnel Decapsulation: The collector Rosen process runs on an independent server, used to decapsulate the tunnel of distributed traffic - - Architecture: System information of the collector's running environment - - Operating System: System information of the collector's running environment - - Control IP: The IP address for communication between the collector and the controller - - Control MAC: The MAC address for communication between the collector and the controller - - Status: Displays the current status of the collector, including unregistered/running/disconnected/disabled - - Exception: When there is an exception during the collector's operation, this column will display a red exclamation mark. Currently supported exception information includes - - Self-check failed: Less than 100MB of remaining space on the log disk - - Self-check failed: Insufficient available memory - - Distribution circuit breaker triggered - - Distribution traffic reached rate limit - - Gateway ARP to distribution point not found - - Gateway ARP to data node not found - - Software Version: Displays the version number of the Trident/Agent software, used for troubleshooting and upgrade indications - - Start Time: Indicates the start time of the collector process - - Controller: Indicates the IP address of the controller from which the collector requests policies (also the destination data node IP address for sending monitoring information). Click to jump to the [Controller List] page. For details, please refer to the [Controller] section - - Controller Sync Time: Displays the last time the collector synchronized cloud platform information with the controller - - Data Node: Indicates the target data node to which the collector sends data. Click to jump to the [Data Node] page. For details, please refer to the [Data Node] section - - Current Access Data Node: When the data node provides services behind SLB, this field displays the real data node IP that the collector is currently requesting - - Data Node Communication Time: Indicates the last communication time between the collector and the data node. Note that the update of this value may have some cache-induced delay - - Actions: Enable/Disable, Delete - - Enable/Disable: Enable or disable the collector - - Delete: Delete unused collectors - -## Group - -Displays information about collector groups in a list form, such as the number of collectors included, the number of disabled collectors, the number of unregistered collectors, etc. - -![02-组](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4c187e7f.png) - -- Displays the number of collectors included in the collector group, the number of disabled collectors, and the number of unregistered collectors in a list form. Click the number to enter the collection page for viewing - - Supports creating new collection groups, registering, disabling, enabling, deleting, and other functions -- Collector Group: Groups hosts/cloud servers of the same type for unified management - - Default: If the user does not customize a group for the collector, they are all classified into the default group - - The default group cannot be modified or deleted, and the collector is subject to the last added group -- Note: For collector groups established by the platform, users do not have permissions for registering collectors, editing, disabling, etc. - -## Configuration - -Displays detailed information about collector groups in a list form, such as CPU limits, memory limits, collection packet rate limits, distribution flow rate limits, distribution circuit breaker thresholds, collection network interfaces, etc. - -![03-配置](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4d1b64aa.png) - -- Click the row: Further displays detailed information about the current collector group, such as resource limits, basic configuration parameters, universal map configuration parameters, packet distribution configuration parameters, basic functions, universal map function switches, packet distribution function switches, etc. - -## Statistics - -Displays current collector-related status data in chart form, such as total collection traffic, total distribution traffic, collector CPU usage, collector memory usage, running environment load, packet loss, cloud server collection traffic, cloud server distribution traffic, etc. - -![04-统计](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4e252f7f.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/12-system/03-data-node.md b/translate/translated/06-guide/01-ee-tenant/12-system/03-data-node.md deleted file mode 100644 index dea40bac..00000000 --- a/translate/translated/06-guide/01-ee-tenant/12-system/03-data-node.md +++ /dev/null @@ -1,30 +0,0 @@ ---- -title: Data Node -permalink: /guide/ee-tenant/system/data-node/ ---- - -> This document was translated by ChatGPT - -# Data Node - -This section presents the configuration information of database-related data tables used on the page in a list format. - -![Data Node](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673ddda82472.png) - -- **Create New Data Table**: Users can customize new data tables based on existing data sources, supporting up to 10 data tables. The creation information is roughly as follows: - - Name: Supports Chinese, English, numbers, and underscores, with a maximum length of 10 characters. - - Data Table Collection: Supports user-created data table collections, presented as database.data_table_collection. - - Options: flow_metrics.vtap_flow*, flow_metrics.vtap_app* - - Original Time Granularity: The original time granularity of the `data table collection`. The system currently supports 1m (one minute) and 1s (one second) by default. - - Time Granularity: The data table will perform aggregation calculations based on the selected time granularity. - - Options: 1h (one hour), 1d (one day) - - Retention Period: Set the data retention period. - - Additive Metric Aggregation: If the original data table uses sum, only sum can be selected; if the original data table uses max/min, only max/min can be selected. - - Non-Additive Metric Aggregation: todo -- **Edit**: Supports modifying the retention period of the data table. -- **Delete**: Only supports deleting user-defined new data sources. -- **Note**: - - The system has 15 default basic data tables, none of which can be deleted. - - Metric data tables (flow*metrics*_) and log data tables (flow*log*_) have a default retention period of 1 week for minute-granularity data and 1 day for second-granularity data. Expired data is automatically deleted. Users can set an appropriate retention period based on this. - - PCAP data is retained for 1 week by default. Data older than 1 week is automatically deleted. Users can set an appropriate retention period based on this. - - Monitoring data in the system is retained for 1 week by default. Data older than 1 week is automatically deleted. Users can set an appropriate retention period based on this. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/12-system/05-operation-log.md b/translate/translated/06-guide/01-ee-tenant/12-system/05-operation-log.md deleted file mode 100644 index bbd3d405..00000000 --- a/translate/translated/06-guide/01-ee-tenant/12-system/05-operation-log.md +++ /dev/null @@ -1,13 +0,0 @@ ---- -title: Operation Log -permalink: /guide/ee-tenant/system/operation-log/ ---- - -> This document was translated by ChatGPT - -# Operation Log - -Logs related to current system and user operations are displayed in a list format. - -- Log Level: Divided into `ERROR`, `WARN`, `INFO` -- Log Type: Divided into `User Login`, `User Operation`, `System Module`, `System Event`, `System Maintenance` \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/13-configuration/01-settings.md b/translate/translated/06-guide/01-ee-tenant/13-configuration/01-settings.md deleted file mode 100644 index 9d7b114e..00000000 --- a/translate/translated/06-guide/01-ee-tenant/13-configuration/01-settings.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -title: Settings -permalink: /guide/ee-tenant/configuration/settings/ ---- - -> This document was translated by ChatGPT - -# Settings - -The settings module supports editing preferences, viewing platform information, and more. - -## Preferences - -Supports configuring the usage preferences of the search box on the page. - -### Search Box Configuration - -The search box configuration only applies to pages under [Events], [Applications], and [Network]. - -![Search Box Configuration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a4bfcef96.png) - -- Follow System Settings: If selected, the configuration is consistent with the default system settings. If you want to adjust the settings yourself, uncheck it. -- **Page Initial Load**: Whether to query data when the page is first loaded - - Do Not Trigger Search: No data on the initial page load. You need to click the [Search] button or add query conditions to display data. - - Search by Default Conditions: `Default system setting`, data is queried and displayed when the page loads. -- **Search Trigger Method**: Set the method to trigger search queries - - Instant Trigger: `Default system setting`, queries are triggered immediately when search conditions change. - - Click [Search] Button to Trigger: When search conditions change, you need to click the [Search] button to trigger the query. -- **Default Search Box Form**: Set the default display form of the search box for `Path` type pages - - Simplified Search: `Default system setting`, for details, please refer to [Resource Search Box](../query/service-search/) - - Unidirectional Path: For details, please refer to [Path Search Box](../query/path-search/) - - Bidirectional Path: For details, please refer to [Path Search Box](../query/path-search/) -- **Default Search Box Content**: The search box can be set to quick search mode - - - Free Search: For details, please refer to [Resource Search Box](../query/service-search/) - - Container Search: For details, please refer to [Resource Search Box](../query/service-search/) - - Process Search: `Default system setting`, for details, please refer to [Resource Search Box](../query/service-search/) - -## Platform Information - -Platform information allows you to view the current system version number, feedback email, vendor information, and more. - -![Platform Information](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645829f09cbb.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-quick-start/01-5w-method.md b/translate/translated/06-guide/01-quick-start/01-5w-method.md new file mode 100644 index 00000000..5cd303ac --- /dev/null +++ b/translate/translated/06-guide/01-quick-start/01-5w-method.md @@ -0,0 +1,312 @@ +--- +title: DeepFlow 5W Fault Diagnosis Method +permalink: /guide/quick-start/5w-method/ +--- + +> This document was translated by ChatGPT + +# Overview + +## Unified Observability Data Lake + +The DeepFlow observability platform aggregates massive amounts of observability data such as metrics, tracing, logging, profiling, and events through eBPF collection and open data interfaces. + +With its AutoTagging label injection technology, DeepFlow can enrich all observability data with detailed text-based labels, including resource tags, business tags, and more — for example, cloud resource information of application instances, container resource information, developer/maintainer/version/commit_id/repository address from CI/CD pipelines, etc. With these labels, you can retrieve all metrics, tracing, logging, profiling, and events related to a specific business or application in one search, display them on a single dashboard, and reach a fault diagnosis conclusion in 3–5 steps of data analysis. + +![Unified Observability Data Lake](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3281c6db18.jpeg) + +## Unified Collaboration Across Teams + +In typical IT system operations, business deployments span multiple availability zones, with complex architectures and numerous components. Operations assurance and fault diagnosis require extensive cross-team communication and collaboration between application, PaaS platform, IaaS cloud, and network teams. The DeepFlow observability platform breaks down data silos, builds data correlations, and provides unified observability and collaboration capabilities for application, PaaS, IaaS, and network operations teams. + +![Unified Collaboration Across Teams](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080866b4b0683ed7b.jpeg) + +## 5W Structured Fault Diagnosis + +![“5W Fault Diagnosis Method”](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024091466e55024e7297.jpeg) + +What is the “**5W Fault Diagnosis Method**”? + +In the DeepFlow observability platform, observability data is reviewed in an orderly manner from macro to micro, progressively answering the following five questions to quickly and effectively identify the root cause of a problem — this is called the “**5W Fault Diagnosis Method**”: + +- **Who** is in trouble? +- **When** is it in trouble? +- **Which** request is in trouble? +- **Where** is the root position? +- **What** is the root cause? + +![5W Fault Diagnosis Process from the Application Perspective](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080866b4b0696e08d.jpeg) + +# Who & When? + +In the DeepFlow platform, either through automatic monitoring or manual analysis of performance between application services, anomalies in application services are detected, answering the **Who** and **When** questions: + +- Which observability object (e.g., a container service, workload, Pod, or a specific path in an IT system) has degraded service quality (errors, slow responses, or timeouts)? +- At what time point or during which time period did the anomaly occur? + +## Where to Start + +Typically, performance analysis of application services begins from the following entry points: + +- `Tracing` - `Resource Analysis`: Entry point for performance metrics analysis of application services (nodes). [Guide link](../ee-tenant/tracing/service-list/) +- `Tracing` - `Path Analysis`: Entry point for performance metrics analysis of application service access paths (edges). [Guide link](../ee-tenant/tracing/service-statistics/) +- `Tracing` - `Topology Analysis`: Entry point for analyzing application service access topology (surfaces). [Guide link](../ee-tenant/tracing/path-topology/) + +## How to Start Searching + +Commonly, you filter and combine conditions such as namespace (`pod_ns`), container service (`pod_service`), workload (`pod_group`), and application protocol (`l7_protocol`) to observe the performance metrics of the target object. + +- Step 1: If you are responsible for an application system deployed in a Kubernetes namespace named “A”, you can use `pod_ns = A` to observe the RED metrics of all application services in that business system. +- Step 2: To further narrow down to a container service named “b” within namespace “A”, add the filter `pod_svc = b`. +- Step 3: To observe only RED metrics for HTTP protocol calls, add the filter `l7_protocol = http`. + +## Which Performance Metrics to Analyze + +In the DeepFlow platform, **RED** metrics (Rate, Error, Duration) are used as the core indicators for evaluating business/application service quality: + +- **Rate** (`Request Rate`) — Number of requests received per unit time, measuring service throughput/load. +- **Error** (`Error Ratio`) — Proportion of requests returning error responses, used to detect service anomalies. Errors are typically categorized into client-side and server-side causes, with server-side errors being the primary focus. +- **Duration** (`Response Latency`) — Time from request to response, used to detect slow responses. Commonly observed statistics include `average response latency`, `P95 response latency`, and `P99 response latency`. + +![RED Metrics Observability in DeepFlow](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b32824d0259.jpeg) + +At this point, you can begin your observability-based diagnosis journey with DeepFlow. + +Of course, DeepFlow also offers more filtering conditions for flexible use in different scenarios, which you can explore over time. + +# Which? + +After answering **Who & When**, the next step is to answer **Which** — which specific application call is abnormal? + +The DeepFlow platform provides a hidden `right sliding panel` for each observability object. By clicking any object in **metric curves** or **metric statistics lists**, the `right sliding panel` will automatically expand. [Guide link](../ee-tenant/tracing/right-sliding-box/) + +The `right sliding panel` offers multiple data observation windows, including `application metrics`, `endpoint list`, `call logs`, and `network metrics`, for analyzing different dimensions of the object. In the `call logs` subpage, you can review all call logs at the anomaly time and filter abnormal calls (errors, slow responses, or timeouts): + +![Retrieving Call Logs in the Right Sliding Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3282b398ec.jpeg) + +# Where? + +After answering **Which**, the next step is to perform distributed tracing on the abnormal call to answer **Where** — finding the root position of errors, slow responses, or timeouts via the distributed tracing flame graph. + +## What is Distributed Tracing + +Based on eBPF technology, DeepFlow innovatively implements zero-intrusion distributed tracing — no need to generate, inject, or propagate TraceIDs to achieve distributed tracing: + +- [Feature usage guide link](../ee-tenant/tracing/call-chain-tracing/) +- [Bilibili video — Understand DeepFlow Distributed Tracing Flame Graph in 3 Minutes](https://www.bilibili.com/video/BV1di421k7JE/) +- [Bilibili video — Understand DeepFlow Distributed Tracing Principles in 3 Minutes](https://www.bilibili.com/video/BV1ZC411E7ad/) + +![DeepFlow Distributed Tracing Diagram](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3282d820d0.jpeg) + +## How to Find the Root Position via the Distributed Tracing Flame Graph + +We use a simplified application service model to illustrate how to quickly locate the **Root Position** with DeepFlow’s distributed tracing flame graph. + +In this scenario, the `Client` sends an `http get` to the `frontend service`, which queries the `DNS service`, accesses the `MySQL database`, makes an `RPC call`, and finally returns an `http response` to the `Client`. + +![Simplified Application Service Model](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328390aa62.png) + +### Application-Level Issues + +**Application Service — Slow IO Thread** + +If there is a significant latency difference between the `POD NIC Span` and the `system Span` of the `frontend service`, it indicates that the `http get` experienced queuing delays when moving from the POD NIC queue to the `frontend service` processing queue. + +This is often caused by busy IO thread scheduling. + +![Flame Graph Example 1 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284a1f3bb.png) + +**Application Service — Slow Work Thread** + +If there is a significant latency difference between the `system Span` receiving the `http get` and the `system Span` sending the `dns query` in the `frontend service`, it indicates internal processing delays. + +This is often caused by busy work thread scheduling. + +![Flame Graph Example 2 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284d7fe39.png) + +**Middleware — Slow DNS Service Response** + +If the `system Span` of the `DNS service` is significantly long, it indicates that the DNS process took too long to query and return the result, directly causing the slow response. + +![Flame Graph Example 3 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328508566e.png) + +**Middleware — Slow MySQL Service Response** + +Similar to DNS, if the `system Span` of the `MySQL service` is significantly long, it indicates that the MySQL process took too long to process and return the result, directly causing the slow response. + +![Flame Graph Example 4 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3285438f40.png) + +**Other Application Service — Slow RPC Service Response** + +Similar to DNS, if the `system Span` of the `RPC service` is significantly long, it indicates that the RPC process took too long to process and return the result, directly causing the slow response. + +![Flame Graph Example 5 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328575653a.png) + +**Client — Slow Process Handling** + +If the `http response` from the `frontend service` reaches the `POD NIC Span` of the `Client` but takes a while to reach the `system Span`, it indicates queuing delays from the POD NIC queue to the `Client` process. + +![Flame Graph Example 6 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3285982a40.png) + +### Network-Level Issues + +**Network Transmission — Slow TCP Connection Establishment** + +If there is a significant latency difference between the `system Span` and the `POD NIC Span` on the `Client` side, it indicates delays before entering the network. + +This often occurs when the `Client` uses short TCP connections, requiring a TCP handshake before sending `http get`. Packet loss or delays during the handshake can cause this. + +![Flame Graph Example 7 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3283e248cf.png) + +**Network Transmission — Slow Intra-Node Transmission on Client Side** + +If there is a significant latency difference between the `POD NIC Span` and the `Node NIC Span` on the `Client` side, it indicates delays in virtual network transmission within the client’s container node. + +![Flame Graph Example 8 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284069e45.png) + +**Network Transmission — Slow Inter-Node Transmission** + +If there is a significant latency difference between the `Node NIC Span` of the `Client` and the `Node NIC Span` of the `frontend service`, it indicates delays in network transmission between container nodes. + +![Flame Graph Example 9 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3284283c8e.png) + +**Network Transmission — Slow Intra-Node Transmission on Server Side** + +If there is a significant latency difference between the `Node NIC Span` and the `POD NIC Span` of the `frontend service`, it indicates delays in virtual network transmission within the server’s container node. + +![Flame Graph Example 10 (Illustration)](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328473bf8b.png) + +# What? + +After answering “**Where is the root position?**” via distributed tracing in DeepFlow, the next step is to perform multi-dimensional data analysis around the **Root Position** to answer “**What**” (**What is the root cause?**): + +- If distributed tracing triage locates the issue to a specific application process, proceed to **application diagnosis** to analyze multiple dimensions of data for that application instance and determine the root cause. +- If distributed tracing triage locates the issue to network transmission, proceed to **network diagnosis** to analyze multiple dimensions of network data and determine the root cause. +- If the issue is related to system performance (e.g., CPU usage, system load, network interfaces), proceed to **system diagnosis** to analyze multiple dimensions of OS data and determine the root cause. + +## Application Diagnosis + +If the **Root Position** is an application instance, you can analyze resource metrics (CPU, memory, disk, etc.), perform OnCPU continuous profiling, OffCPU continuous profiling, memory profiling, application metrics analysis, and application log retrieval in DeepFlow to find the **Root Cause** inside the application. + +### Application Instance Resource Metrics Analysis + +DeepFlow can integrate and unify observation of compute resource metrics for container Pods/Containers, quickly determining if container resources are the root cause at the anomaly time. + +Entry: `Metrics` - `Container`. [Guide link](../ee-tenant/metrics/container/) + +**Pod Status List Observation** + +![Pod Status Observation Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b32862b4f18.jpeg) + +**Container Metrics Detail Observation** + +![Container Metrics Observation Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328672066c.png) + +### OnCPU Continuous Profiling + +DeepFlow can continuously profile OnCPU usage to find CPU hotspot functions in application processes. + +Entry: `Profiling` - `Continuous Profiling`. [Guide link](../ee-tenant/profiling/continue-profile/) + +![OnCPU Continuous Profiling Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3286e73055.png) + +### OffCPU Continuous Profiling + +DeepFlow can continuously profile OffCPU usage to find blocking functions caused by IO waits, locks, etc. + +Entry: `Profiling` - `Continuous Profiling`. [Guide link](../ee-tenant/profiling/continue-profile/) + +![OffCPU Continuous Profiling Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3288a5dc39.png) + +### Memory Profiling + +DeepFlow can profile memory usage to find memory hotspot functions in application processes. + +Entry: `Profiling` - `Continuous Profiling`. [Guide link](../ee-tenant/profiling/continue-profile/) + +### Application Metrics Analysis + +DeepFlow can integrate and analyze application-exposed metrics to find the root cause inside the program. + +Entry: `Metrics` - `Metrics Viewing`. [Guide link](../ee-tenant/metrics/metrics-viewing/) + +### Application Log Analysis + +DeepFlow can integrate and analyze application logs to find the root cause inside the program. + +Entry: `Log` - `Log`. [Guide link](../ee-tenant/log/log/) + +## System Diagnosis + +### File IO Event Analysis + +When a `system Span` is identified as the Root Position in distributed tracing, you can instantly retrieve the list of slow file IO events associated with that Span to determine if file IO performance is the root cause. + +Entry 1: `Tracing` - `Call Chain Tracing` - `IO Events` +Entry 2: `Right Sliding Panel` - `File Read/Write Events`. [Guide link](../ee-tenant/tracing/right-sliding-box/) +Entry 3: `Tracing` - `File Reading and Writing`. [Guide link](../ee-tenant/tracing/file-reading-and-writing/) + +![File IO Event Analysis Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3287b2741a.png) + +### K8s Resource Change Event Analysis + +Analyze the list of K8s resource change events at the anomaly time for the Root Position to determine if container creation/destruction is the root cause. + +Entry 1: `Right Sliding Panel` - `Resource Change Events`. [Guide link](../ee-tenant/tracing/right-sliding-box/) +Entry 2: `Resources` - `Change Events`. [Guide link](../ee-tenant/resources/resource-changes/) + +![K8s Resource Change Event Analysis Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3287e6aa0e.jpeg) + +### System Metrics Analysis + +DeepFlow can integrate and unify observation of system metrics for cloud servers and container nodes, quickly determining if system resources are the root cause at the anomaly time. + +![Host Metrics List Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b32896e4f7a.png) + +### System Log Analysis + +DeepFlow can integrate and analyze system logs to find the root cause inside the system. + +## Network Diagnosis + +### Network Metrics Analysis + +When a `network Span` is identified as the Root Position in distributed tracing, you can instantly retrieve the `network performance` of the associated application call to determine delays in TCP handshake, TLS handshake, data exchange, and system response, identifying the key reason for slow network transmission: + +- `TCP Connection Delay` — Delay during TCP three-way handshake +- `TLS Connection Delay` — Delay during TLS handshake +- `Average Data Delay` — Delay from request Data to response Data (average over multiple occurrences) +- `Average System Delay` — Delay from request Data to ACK response (average over multiple occurrences) +- `Average Client Wait Delay` — Delay from last ACK or response Data to next request (average over multiple occurrences) + +![Network Metrics Analysis in Distributed Tracing Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328833d96a.png) + +### Flow Log Analysis + +If network performance analysis cannot fully determine the root cause, you can further `view flow logs` to examine detailed TCP session data for retransmissions, zero window, TCP RST, etc. + +![Key Information in Flow Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328a51deb5.jpeg) + +### TCP Sequence Analysis + +If flow logs still cannot fully determine the root cause, you can view the `TCP sequence diagram` corresponding to the flow logs to examine packet interaction sequences and timing differences, identifying anomalies and finding the root cause. + +![TCP Sequence Diagram Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b3289a9c9aa.png) + +**TCP Sequence Diagram Root Cause Analysis Case** + +Fault: The server did not receive any new business requests within 15 seconds after sending a response packet to the client, triggering a timeout and closing the TCP connection. However, 53ns later, a new request packet arrived from the client. Since the TCP connection was already closed, the server could not process it and sent an RST to notify the client to stop sending requests. + +Impact: The last application request had no response. + +Solution: Increase the server’s TCP connection timeout (longer than the client’s) and ensure the client actively closes the TCP connection after each business process. + +![TCP Sequence Diagram Fault Analysis Case](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024080766b328a3b7ad0.jpeg) + +### Network Device Metrics Analysis + +DeepFlow can also integrate network device operational metrics via Telegraf. When distributed tracing locates the Root Position in the physical network, you can analyze network device metrics to find the root cause. + +# Summary + +By answering the five questions (**Who / When / Which / Where / What**), we can identify the root cause of a problem and provide targeted solutions to eliminate the fault. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/01-query/01-overview.md b/translate/translated/06-guide/02-ee-tenant/01-query/01-overview.md new file mode 100644 index 00000000..6b57309b --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/01-query/01-overview.md @@ -0,0 +1,23 @@ +--- +title: Search +permalink: /guide/ee-tenant/query/overview/ +--- + +> This document was translated by ChatGPT + +# Search + +DeepFlow is a highly automated observability platform that provides application and network metrics, events, tracing, and log data. With its AutoTagging capability, it injects unified attribute tags into all observability data. When dealing with massive volumes of data, DeepFlow not only offers productized analytical capabilities but also supports fast search. As an efficient information retrieval tool, the search box enables quick and accurate lookup and filtering in the face of large-scale data. This chapter focuses on how to use the DeepFlow search box. + +The DeepFlow search box can be categorized into four main types, and this chapter will provide a detailed explanation of each type and its common application scenarios. + +- [Resource Search Box](./service-search/) +- [Path Search Box](./path-search/) +- [Log Search Box](./log-search/) +- [Metric Search Box](./metric-search/) + +In addition to defining the search box, DeepFlow also enhances it with quick search capabilities and search-related configurations. + +- [Search Snapshots](./history/) +- [Left-Side Quick Filter](./left-quick-filter/) +- [Search Box Settings](../configuration/settings/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/01-query/02-service-search.md b/translate/translated/06-guide/02-ee-tenant/01-query/02-service-search.md new file mode 100644 index 00000000..7b52b071 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/01-query/02-service-search.md @@ -0,0 +1,117 @@ +--- +title: Resource Search Box +permalink: /guide/ee-tenant/query/service-search/ +--- + +> This document was translated by ChatGPT + +# Resource Search Box + +The `Resource Search Box` is used in Application - Resource Analysis, Network - Resource Analysis, and Network - Resource Inventory. + +![01-Resource Search Box](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664ac73e1e086.png) + +- **① Search Snapshot**: Refer to the [Search Snapshot](./history/) section for details +- **② Search Input Mode**: Allows switching between different search input modes, currently including Free Search, Container Search, and Process Search. See the following sections for details. + +## Free Search + +![02-Free Search](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405156644260e09259.png) + +- **① Search Condition Input Box**: Supports Chinese and English auto-completion, and supports using Tags from the data table as search conditions +- **② Clear Search Conditions**: Clears the `Search Condition Input Box` +- **③ Switch Primary Group**: Resource grouping, corresponding to `Resource` in the feature interface +- **④ Switch Secondary Group**: Other groupings, corresponding to `Group Attributes` in the feature interface + +In the `Search Condition Input Box`, each complete search condition is called a `Search Tag`. The following explains in detail how to manage `Search Tags`. + +![03-Search Tag](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa57a56f.png) + +![04-Operators](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa702aed.png) + +![05-Candidates](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c50ecc63c1.png) + +- **① Tag Name**: Supports querying Tags in the data table. For detailed descriptions, see `Database Fields` + - Supports Chinese and English auto-completion + - Hover over the Tag name to view detailed information + - Semantics: Different Tags are connected with `and`. The same Tag uses different logical operators depending on the `operator`: + - a: `=` , `:` , `~` are connected with `or` + - b: `!=` , `!:` , `!~` are connected with `and` + - c: `>=` , `<=` , `>` , `<` are connected with `and` + - After connecting a/b/c, they are further connected with `and`. For example: + `Search Condition Input Box: server_port > 20, server_port < 80, server_port != 44, server_port != 45` + The effective condition is `(server_port > 20 and server_port < 80) and (server_port != 44 and server_port != 45)` +- **② Operator**: Currently supports exact match, fuzzy match, and regex match + - Exact Match: Corresponds to `=` , `!=` , `>=` , `<=` operators. For `resource type` Tags, exact match is based on resource ID; for others, it matches exactly as entered + - Fuzzy Match: Corresponds to `:` , `!:` operators. String matching supports `*` wildcard. For example, `*123*` matches all strings containing `123`, while `123` matches only strings exactly equal to `123` + - Regex Match: Corresponds to `~` , `!~` operators. Performs string regex matching +- **③ Tag Value**: Select or directly enter the value to filter + - NULL: Null value, usually used with `!=` to mean filtering out `all` + - **⑦ Table Filtering**: When candidate items have duplicate names or multiple selections are needed, use `Table Filtering` to precisely locate resources +- **④ Disable**: Disables the search condition corresponding to the current `Search Tag` +- **⑤ Edit**: Edits the search condition corresponding to the current `Search Tag` +- **⑥ Delete**: Deletes the current `Search Tag` + +## Container Search + +The container search mode fixes commonly used resource Tags in container scenarios as dropdown menus, making it easy to quickly filter container resources. + +![06-Container Search](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240515664425a2b3c16.png) + +- **① Container Resource Dropdown**: Click the dropdown to quickly select the container resource to filter. The options in the following dropdowns can be linked to the previous selection. +- **② Search Condition Input Box**: See the `Free Search` section above for details +- **③ Collapse Search Condition Input Box**: Click to quickly collapse the `Search Condition Input Box` +- **④ Switch Group**: Quickly switch container resource Tags + +## Process Search + +The process search mode is similar to container search, but fixes commonly used process-related Tags as dropdown menus to quickly filter process resources. + +![07-Process Search](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240515664426411eea4.png) + +# Application Scenarios + +## View the service performance of a specific workload + +- Feature Page: Application - Metrics +- Search Tag: pod_ns = gcp-microservices-demo +- Primary Group: auto_service +- Secondary Group: -- + +![05-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa039078.png) + +## View the performance of a specific workload in a specific namespace + +- Feature Page: Application - Metrics +- Search Tag: pod_ns = gcp-microservices-demo, pod_group : loadgenerator +- Primary Group: auto_service +- Secondary Group: -- + +![06-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa17b7c6.png) + +## View the top 5 cloud servers by traffic + +- Feature Page: Network - Services +- Search Tag: None +- Primary Group: chost +- Secondary Group: -- + +![07-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa2642e9.png) + +## View the top 5 server ports by traffic for a specific cloud server + +- Feature Page: Network - Services +- Search Tag: role = server +- Primary Group: chost +- Secondary Group: server_port + +![08-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa2adfda.png) + +## View the network performance of a specific port on a specific cloud server + +- Feature Page: Network - Services +- Search Tag: role = server, chost = cn-chengdu.172.16.0.196, server_port = 22 +- Primary Group: chost +- Secondary Group: server_port + +![09-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4fa44b491.png) \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/01-query/03-path-search.md b/translate/translated/06-guide/02-ee-tenant/01-query/03-path-search.md similarity index 66% rename from translate/translated/06-guide/01-ee-tenant/01-query/03-path-search.md rename to translate/translated/06-guide/02-ee-tenant/01-query/03-path-search.md index 7170dd24..92aefee8 100644 --- a/translate/translated/06-guide/01-ee-tenant/01-query/03-path-search.md +++ b/translate/translated/06-guide/02-ee-tenant/01-query/03-path-search.md @@ -12,71 +12,71 @@ The `Path Search Box` is used in Application - Path Analysis/Topology Analysis, ![00-Path Search Box](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032065faac6eb6f17.png) - **①/②/③/④/⑦**: For detailed operation instructions, please refer to [Resource Search Box](./service-search/) -- **⑤ Search Mode**: You can switch between `Simplified Mode`, `Unidirectional Path`, and `Bidirectional Path` modes, and use the `Path Filtering` capability to query the required path data. - - Simplified Search: The entered `search tags` will be used as conditions for both `client` and `server`. For details, please refer to the [Simplified Mode] section. - - Unidirectional Path: The entered `search tags` will be used as conditions for either the `client` or the `server`. For details, please refer to the [Unidirectional Path] section. - - Bidirectional Path: The entered `search tags` do not specify the query direction. For details, please refer to the [Bidirectional Path] section. -- **⑥ Path Filtering**: Supported only in `Simplified Search` mode. For details, please refer to the [Simplified Mode] section. +- **⑤ Search Mode**: You can switch between `Simplified Mode`, `Unidirectional Path`, and `Bidirectional Path` modes. Combined with the `Path Filter` capability, you can query the required path data. + - Simplified Search: The entered `search tags` will be used as conditions for both `client` and `server` queries. For details, please refer to the **Simplified Mode** section. + - Unidirectional Path: The entered `search tags` will be used as conditions for either the corresponding `client` or `server`. For details, please refer to the **Unidirectional Path** section. + - Bidirectional Path: The entered `search tags` do not specify a query direction. For details, please refer to the **Bidirectional Path** section. +- **⑥ Path Filter**: Supported only in `Simplified Search` mode. For details, please refer to the **Simplified Mode** section. ## Simplified Mode ![01-Simplified Mode](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032065fab1ff0d41a.png) -The Path Search Box can switch between `Simplified Mode`, `Unidirectional Path`, and `Bidirectional Path` modes, and use the `Path Filtering` capability to query the required path data. +The Path Search Box can switch between `Simplified Mode`, `Unidirectional Path`, and `Bidirectional Path` modes. Combined with the `Path Filter` capability, you can query the required path data. - **①/②/③/④**: For detailed operation instructions, please refer to [Resource Search Box](./service-search/) - **⑤ Search Mode**: Click to switch between `Simplified Search` and `Path Search` modes. -- **⑤ Path Filtering**: Supports selecting the query path and supports querying three types of paths, only supported in `Simplified Search` mode. +- **⑤ Path Filter**: Supports selecting the path to query and supports querying three types of paths. Only available in `Simplified Search` mode. - Intra-service Path: Paths between services or resources. - - Inter-service Path: Paths between a service or resource and other services or resources. - - WAN Path: Paths between a service or resource and the WAN. + - Inter-service Path: Paths between a service/resource and other services/resources. + - WAN Path: Paths between a service/resource and the wide area network. ## Unidirectional Path ![02-Unidirectional Path](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032065fab3eccc2de.png) - **①/②/③/④**: For detailed operation instructions, please refer to [Resource Search Box](./service-search/) -- **⑤ Search Mode**: You can switch the search mode. For details, please refer to the [Path Search Box] section. -- **⑥ Swap Direction**: Click to quickly swap the search conditions of `client` and `server`, only supported in `Unidirectional Path` mode. +- **⑤ Search Mode**: You can switch the search mode. For details, please refer to the **Path Search Box** section. +- **⑥ Swap Direction**: Click to quickly swap the search conditions of `client` and `server`. Only available in `Unidirectional Path` mode. ## Bidirectional Path ![03-Bidirectional Path](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032065fab5961fd4e.png) - **①/②/③/④**: For detailed operation instructions, please refer to [Resource Search Box](./service-search/) -- **⑤ Search Mode**: You can switch the search mode. For details, please refer to the [Path Search Box] section. +- **⑤ Search Mode**: You can switch the search mode. For details, please refer to the **Path Search Box** section. # Application Scenarios -## View the Call Topology of All Services in a Namespace +## View the call topology of all services within a namespace - Function Page: Application - Topology Analysis - Data Table: Metrics (minute-level) --- -- Search Tags: pod_ns = gcp-microservices-demo +- Search Tag: pod_ns = gcp-microservices-demo - Path: Intra-service - Primary Group: auto_service - Secondary Group: observation_point ![04-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f8b00145.png) -## View the Application Performance of All Paths of a Service +## View the application performance of all paths for a service - Function Page: Application - Path Overview - Data Table: Metrics (minute-level) --- -- Search Tags: pod_ns = gcp-microservices-demo, pod_service = productpageservice +- Search Tag: pod_ns = gcp-microservices-demo, pod_service = productpageservice - Path: Intra-service, Inter-service, WAN - Primary Group: auto_service - Secondary Group: observation_point ![05-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f8b659a4.png) -## View the Application Performance of a Service Accessing an External MySQL +## View the application performance of a service accessing an external cloud MySQL - Function Page: Application - Path Overview - Data Table: Call Logs @@ -84,48 +84,48 @@ The Path Search Box can switch between `Simplified Mode`, `Unidirectional Path`, --- - Direction: Client -- Search Tags: pod_service = finaxxx, l7_protocol = MySQL +- Search Tag: pod_service = finaxxx, l7_protocol = MySQL - Primary Group: auto_service - Secondary Group: observation_point --- - Direction: Server -- Search Tags: ip = 8.x.x.x (address of the external MySQL) +- Search Tag: ip = 8.x.x.x (address of the external cloud MySQL) - Primary Group: auto_service - Secondary Group: observation_point ![06-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f8c96f16.png) -## View the Application Performance of Two Workloads Interacting at the POD Level +## View application performance at POD granularity for mutual access between two workloads - Function Page: Application - Path Overview - Data Table: Metrics (minute-level) --- -- Search Tags: pod_group = recommendationservice, pod_group = productcatalogservice +- Search Tag: pod_group = recommendationservice, pod_group = productcatalogservice - Path: Intra-service - Primary Group: auto_service - Secondary Group: observation_point ![07-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f8d313a2.png) -## View the Performance Data of a Path Corresponding to a Domain Name +## View performance data for paths corresponding to a specific domain name - Function Page: Application - Path Overview - Data Table: Call Logs --- -- Search Tags: request_domain : hotels.travel-agency:8000 +- Search Tag: request_domain : hotels.travel-agency:8000 - Path: Intra-service, Inter-service, WAN - Primary Group: auto_service - Secondary Group: observation_point, request_domain ![08-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f8de4a50.png) -## View the Top 5 Other Resources Accessing a Cloud Server +## View the top 5 other resources accessing a specific cloud server by traffic - Function Page: Network - Path Analysis - Data Table: Metrics (minute-level) @@ -133,20 +133,20 @@ The Path Search Box can switch between `Simplified Mode`, `Unidirectional Path`, --- - Direction: Client -- Search Tags: chost != cn-zhxxx +- Search Tag: chost != cn-zhxxx - Primary Group: auto_service - Secondary Group: observation_point --- - Direction: Server -- Search Tags: chost = cn-zhxxx +- Search Tag: chost = cn-zhxxx - Primary Group: chost - Secondary Group: observation_point ![09-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f8eb246e.png) -## View the Top 5 WAN IPs Accessing a Cloud Server's Service Port +## View the top 5 WAN IPs by traffic for a specific service port on a cloud server - Function Page: Network - Path Overview - Data Table: Flow Logs @@ -154,20 +154,20 @@ The Path Search Box can switch between `Simplified Mode`, `Unidirectional Path`, --- - Direction: Client -- Search Tags: is_internet = yes +- Search Tag: is_internet = Yes - Primary Group: auto_service - Secondary Group: observation_point --- - Direction: Server -- Search Tags: chost = cn-zhxx, server_port = 80 +- Search Tag: chost = cn-zhxx, server_port = 80 - Primary Group: chost - Secondary Group: observation_point, server_port ![10-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f8f9ac67.png) -## View the Network Performance of a Client (POD) Accessing a Server (POD) +## View the network performance of a client (POD) accessing a server (POD) - Function Page: Network - Path Overview - Data Table: Metrics (minute-level) @@ -175,14 +175,14 @@ The Path Search Box can switch between `Simplified Mode`, `Unidirectional Path`, --- - Direction: Client -- Search Tags: pod = nginx-xxxx +- Search Tag: pod = nginx-xxxx - Primary Group: pod - Secondary Group: observation_point --- - Direction: Server -- Search Tags: pod = bohriu-xxxx +- Search Tag: pod = bohriu-xxxx - Primary Group: pod - Secondary Group: observation_point diff --git a/translate/translated/06-guide/02-ee-tenant/01-query/04-log-search.md b/translate/translated/06-guide/02-ee-tenant/01-query/04-log-search.md new file mode 100644 index 00000000..c49945a1 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/01-query/04-log-search.md @@ -0,0 +1,79 @@ +--- +title: Log Search Box +permalink: /guide/ee-tenant/query/log-search/ +--- + +> This document was translated by ChatGPT + +# Log Search Box + +The `Log Search Box` is used in Application - Call Logs / Distributed Tracing and Network - Flow Logs. + +Compared with the `Path Search Box`, the `Log Search Box` only lacks the `grouping` capability. For detailed operation instructions, please refer to [Path Search Box](./path-search/) + +![00-Log Search Box](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f5e7a6cd.png) + +# Application Scenarios + +## View calls with exceptions for a specific service + +- Feature Page: Application - Call Logs + +--- + +- Service Set: S1 +- Search Tags: pod_service = frontend-external, response_status != Normal +- Path: Intra-Service, Inter-Service, WAN +- Direction: Bidirectional + +![01-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f61ad6e0.png) + +## View MySQL calls for a specific service + +- Feature Page: Application - Call Logs + +--- + +- Service Set: S1 +- Search Tags: pod_service = cars, l7_protocol = MySQL +- Path: Intra-Service, Inter-Service, WAN +- Direction: Bidirectional + +![02-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f60c6540.png) + +## View flow logs for a specific five-tuple + +- Feature Page: Network - Flow Logs + +--- + +- Service Set: S1 +- Search Tags: pod = insurances-v1-d895774d6-26wf7, client_port = 46168, protocol = TCP +- Path: Intra-Service +- Primary Group: pod +- Secondary Group: observation_point +- Direction: Client + +--- + +- Service Set: S2 +- Search Tags: pod = mysqldb-v1-5cc78df8d-fwrn4, server_port = 3306 +- Path: Intra-Service +- Primary Group: pod +- Secondary Group: observation_point +- Direction: Server + +![03-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f601adf9.png) + +## View flow logs with connection establishment exceptions for a specific POD + +- Feature Page: Network - Flow Logs + +--- + +- Service Set: S1 +- Search Tags: pod = frontend-97cc49c74-qs6wh, close_type = Connection Establishment - Client ACK Missing, close_type = Connection Establishment - Server SYN Missing, close_type = Connection Establishment - Client Port Reuse, close_type = Connection Establishment - Server Direct Reset, close_type = Connection Establishment - Other Server Reset +- Path: Intra-Service, Inter-Service, WAN +- Direction: Bidirectional + +![04-Query Result](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b087b6e86.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/01-query/05-metric-search.md b/translate/translated/06-guide/02-ee-tenant/01-query/05-metric-search.md new file mode 100644 index 00000000..84d2d5bf --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/01-query/05-metric-search.md @@ -0,0 +1,23 @@ +--- +title: Metric Search Box +permalink: /guide/ee-tenant/query/metric-search/ +--- + +> This document was translated by ChatGPT + +# Metric Search Box + +Currently, both the `Metrics Page` and `Chart - Edit - Search Conditions` use the `Metric Search Box`. + +![Metric Search Box](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c4f741fb51.png) + +- **① Database**: The database where the metric resides, such as Application, Network, Event, or Prometheus. +- **② Data Table**: The data table where the metric resides, for example, the `Metrics (Minute Level)` or `Metrics (Second Level)` tables under the `Application` database. +- **③-⑦**: Please refer to the sections [Service Search Box](./service-search/), [Path Search Box](./path-search/), and [Log Search Box](./log-search/). +- **⑧ Switch to PromQL Input Box**: Click the button to switch between `Simplified Search` and `PromQL Search` modes. +- **⑨ Metric Dropdown**: Select the metric you want to view. Note: You must select one metric. +- **⑩ Operator Dropdown**: Select an aggregation operator. For detailed explanations of operators, please refer to the document [Calculation Logic of Metric Operators](../../../features/universal-map/metrics-and-operators/#%E8%81%9A%E5%90%88%E7%AE%97%E5%AD%90). +- **⑪ Secondary Operator**: Select a secondary operator. +- **⑫ Disable/Enable**: If a metric is disabled, it will not be queried; if enabled, the query will be executed. +- **⑬ Add Metric**: Supports adding multiple metrics, each corresponding to a line chart. +- **⑭ Add Query**: Supports adding multiple query conditions. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/01-query/06-history.md b/translate/translated/06-guide/02-ee-tenant/01-query/06-history.md new file mode 100644 index 00000000..c3ec768f --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/01-query/06-history.md @@ -0,0 +1,51 @@ +--- +title: Search Snapshot +permalink: /guide/ee-tenant/query/history/ +--- + +> This document was translated by ChatGPT + +# Search Snapshot + +The Search Snapshot feature records the user's past search-related criteria, helping you save the current page's query conditions, query time, and chart configuration settings. It also allows you to quickly select a search snapshot from a dropdown menu to apply it to the page, as well as share search snapshots, set a default loading page, and other functions. + +Next, we will introduce how to use the Search Snapshot feature. + +## Basic Introduction + +![00-Basic Introduction](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d6263c60c9.png) + +- **① Search Snapshot Dropdown:** Displays all saved search criteria for the current page in a dropdown menu, and also supports managing search snapshots. For detailed usage, please refer to **Search Snapshot Dropdown**. +- **② Save Search Criteria:** Click to save the search criteria, time, and other information on the current page. For details, please refer to **Save Search Criteria**. + +## Search Snapshot Dropdown + +![01-Dropdown](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d626419381.png) + +The Search Snapshot bar consists of a search bar, dropdown list, and description box. + +- Search Bar: Allows you to search for search snapshot names, supporting both Chinese and English suggestions. + - Click the search bar to open the dropdown list, which displays the search snapshots saved for the current page. +- Dropdown List: Displays the search criteria records saved by the user on the current page, as well as search criteria for the current page shared by other users. + - Also supports starring, editing, and other operations on search snapshots. For details, please refer to the **Manage Search Snapshots** section. +- Description Box: When hovering over a search snapshot, the description box displays related information. + - Shows the search snapshot's name, description, number of searches, permissions, source account, creation time, and more. + +## Save Search Criteria + +![02-Save Search Criteria](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d6264dba0a.png) + +Users can save the search criteria on the current page. Click the `Save` icon to edit the name and description of the saved snapshot. It also supports remembering the search time range and the configuration of Panels. + +## Manage Search Snapshots + +![03-Manage Search Snapshots](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230922650d6265ca207.png) + +- **① Star:** Marks the search snapshot as important, displaying it at the top of the dropdown list and table to help users find it faster. + - Click again to remove the star. +- **② Edit:** Modify the `name` or `description` of the search snapshot. +- **③ Share:** Allows sharing the search snapshot with one or more specified users, with the option to assign `read-only` or `read-write` permissions. Also shows the number of times it has been shared. +- **④ Query:** Opens the search snapshot criteria in a new page, and shows the number of times it has been queried. +- **⑤ Set as Default Loading Page:** Click the icon to set the search snapshot criteria as the default loading page for the current page. + - Once set successfully, the icon will be highlighted, and the setting will take effect when re-entering from other pages. +- **⑥ Delete:** Deletes the search snapshot. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/01-query/07-left-quick-filter.md b/translate/translated/06-guide/02-ee-tenant/01-query/07-left-quick-filter.md new file mode 100644 index 00000000..38d5a7d2 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/01-query/07-left-quick-filter.md @@ -0,0 +1,32 @@ +--- +title: Left-Side Quick Filter +permalink: /guide/ee-tenant/query/left-quick-filter/ +--- + +> This document was translated by ChatGPT + +# Left-Side Quick Filter + +The left-side quick filter feature supports fast filtering of **tag** and **metric** fields. The following uses the Path Overview page as an example to demonstrate how to use the left-side filter. + +![00-Left-Side Quick Filter](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a9fb1183e5.png) + +The left-side quick filter allows you to filter table data on the page by specific fields, enabling quick searches and improving query efficiency. Currently, the left-side quick filter only supports certain fields, with more fields to be gradually supported in the future. + +## Usage Guide + +Click the `Quick Filter` button in the upper left corner to expand a panel on the left side of the page showing the available filter fields. Hovering over a data item displays explanations for the data and option values. When the left-side quick filter is active, the query conditions in the page’s service search bar are synchronized. + +![01-Usage Guide](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a9fb139c2f.png) + +- The left-side quick filter panel is open by default. +- **Operation Instructions:** + - When hovering over a data item or checkbox, a tooltip will indicate the state after clicking at that position. + - State descriptions: + - **Select All**: By default, all options under each field are selected, and no filtering is applied to the page query. + - **Select Only This Item**: Clicking the row of an option means only the current option value for that field will be used as the query condition. + - **Deselect**: Clicking the row of an option cancels the **Select Only This Item** state and restores **Select All**. + - **Toggle State**: Clicking the checkbox selects or deselects the option. + - **Selected**: The checkbox is checked, meaning the option will be included in the query. + - **Deselected**: The checkbox is unchecked, meaning the option will be excluded from the query. + - **Clear Filter**: Clicking the clear filter icon in the upper right corner of the data panel clears the value filter for that field and restores **Select All**. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/02-dashboard/01-overview.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/01-overview.md similarity index 54% rename from translate/translated/06-guide/01-ee-tenant/02-dashboard/01-overview.md rename to translate/translated/06-guide/02-ee-tenant/02-dashboard/01-overview.md index 0dcb3313..e02ddbbd 100644 --- a/translate/translated/06-guide/01-ee-tenant/02-dashboard/01-overview.md +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/01-overview.md @@ -7,10 +7,10 @@ permalink: /guide/ee-tenant/dashboard/overview/ # Overview -A Dashboard is composed of one or a group of charts. DeepFlow offers a rich variety of charts, allowing you to easily customize visualization panels for various scenarios. -We provide the following instructions to help you quickly understand and use the DeepFlow Dashboard features. +A dashboard consists of one or more charts. DeepFlow offers a rich variety of charts, allowing you to easily customize visualization panels for different scenarios. +We provide the following instructions to help you quickly understand and use the DeepFlow dashboard features. - [Dashboard List](./list/) - [Dashboard Details](./use/) -- [Add Panel](./add-panel/) +- [Add Chart](./add-panel/) - [Variable Template](./variable-template/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/02-list.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/02-list.md new file mode 100644 index 00000000..05d63a86 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/02-list.md @@ -0,0 +1,26 @@ +--- +title: Dashboard List +permalink: /guide/ee-tenant/dashboard/list/ +--- + +> This document was translated by ChatGPT + +# Dashboard List + +The Dashboard List page displays all dashboards created by the current user along with some basic operations. + +![overview.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240514664305b709a65.png) + +- Dashboards are divided into two categories: Custom Dashboards and Built-in Dashboards + - **Custom Dashboards:** Show dashboards created by the current account and those editable within the team or organization + - **Built-in Dashboards:** Provided by DeepFlow to visualize relevant system information; cannot be edited or deleted +- **① Create Dashboard:** Click "Create Dashboard", enter a name for the new dashboard, and it will be created. You can also add a description or notes as needed +- **② Import Dashboard:** Click "Import Dashboard" to define a name and select a JSON file to import. Note: Currently, only JSON dashboard files exported from DeepFlow are supported. For details, see `Export` +- **③ Search:** Enter any string in the search bar, such as name, team, description, creator, or last modified time, to match and filter the list +- **④ Settings:** Configure how column widths are displayed, such as evenly distributed or adjusted based on content +- **⑤ Delete:** Supports bulk deletion of all selected dashboards +- **⑥ Export:** Supports bulk export of all selected dashboards +- **⑦ Star:** Starred dashboards will be displayed with priority +- **⑧ More:** Groups together the Edit, Export, and Delete functions + - **Edit:** Allows modification of the dashboard name and description + - **Export:** Supports exporting dashboards in JSON format \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/03-use.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/03-use.md new file mode 100644 index 00000000..e293e461 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/03-use.md @@ -0,0 +1,72 @@ +--- +title: Dashboard Details +permalink: /guide/ee-tenant/dashboard/use/ +--- + +> This document was translated by ChatGPT + +# Dashboard Details + +The dashboard details page displays user-defined visualization panels. + +![00-Dashboard Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031165eec281b77db.png) + +- **① Dashboard Dropdown:** The dropdown options are dashboard names. Select a name to quickly switch dashboards. +- **② Query Area:** Supports one-click switching of the chart query area, and can also switch the query area directly on the chart. +- **③ Time Picker:** Allows customization of the time range for the visualization panel data. For details, see the **Time Picker** section. +- **④ Time Interval:** Allows selection of the time granularity for data aggregation. Note: Aggregation granularity applies only to time series charts: TOP N line chart / line chart / trend analysis chart. + - Seconds: 1, 5, 10, 30s + - Minutes: 1, 5, 10, 30m + - Hours: 1, 3, 6, 12h + - Days: 1, 7d +- **⑤ Manual Refresh:** Click the refresh button to update data in real time. +- **⑥ Auto Refresh:** Auto refresh is disabled by default. You can enable it to refresh automatically every 1m or 5m. +- **⑦ Add Chart:** Supports adding charts and groups. For details, see the **Add Chart** section. +- **⑧ Full Screen:** Click the button to display the current dashboard in full screen. Press `Esc` to exit full screen. +- **⑨ Save:** After modifying the dashboard's time range, time interval, topology position, chart configuration, template variable values, query area, etc., click the save button to save the changes. To save as a copy, check the "Save As" option. +- **⑩ Settings:** In the settings menu, you can export, delete, manage template variables, and perform other operations on the dashboard. For details, see the **Settings** section. + +## Time Picker + +The time picker supports viewing historical dashboard data using either an `absolute time` range or a `relative time` range. + +![01-Time Picker](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031165eec28050664.png) + +- Absolute Time: + - Supports selecting a time range from the calendar. + - Supports manually entering a time range in the `YYYY-MM-DD HH-mm-ss` format. +- Relative Time: + - Supports quick shortcuts to select relative time ranges: + - Last 5m, 15m, 30m, 6h, 1d, 7d, 30d, etc. +- Supports entering relative time using the `now` keyword: + - `now`: The exact time corresponds to the current date and time down to the second. + - `now/d`: Represents "today". If used as the start time, it corresponds to 00:00:00 today; if used as the end time, it corresponds to 23:59:59 today. + - `now-$num d`: Represents the last *num* days. The exact time is the current time minus the specified number of days. *num* can be an integer from 1 to 100. +- Search Snapshot Records: Stores historical search times for quick access. + +## Add Chart + +There are two ways to add charts to a dashboard: add charts directly within the dashboard, or add charts from an external page to the dashboard. + +![02-Add Chart in Dashboard](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240514664327cf0b5d6.png) + +- Click the `Add Chart` button to add line charts, bar charts, pie charts, overview charts, traffic topology, distribution charts, tables, text, and more. Grouping is also supported. +- For adding charts from an external page, see the **[Add Chart](./add-panel/)** section. + +## Settings + +The settings button in the dashboard page provides a variety of functions to help users better utilize the dashboard. + +![03-Settings](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031165eec3a58224c.png) + +- Set Global Data Table: Quickly switch the data table referenced by charts in the dashboard. +- Manage Template Variables: Quickly change chart search conditions. For details, see the **[Template Variables](./variable-template/)** section. +- Enable/Disable Tip Sync: When enabled, you can view data at the same time point across all time series-related charts. + - Time series-related charts include: line charts, trend analysis charts. +- Create Module: Organize charts into modules. Modules can be collapsed or expanded as needed, and you can rename or delete modules from the module bar. +- Switch Fill Method: When data is missing at a certain time point, choose a fill method to handle it: + - Fill 0: Fill with 0 at the current time point. + - Fill null: Leave the current time point empty. + - Fill none: Remove the current time point. +- Switch Tile/Stack: Quickly toggle between tiled and stacked display for all time series-related charts in the dashboard details page. +- Full/Default Name Display: Quickly toggle the name display mode for all charts in the dashboard details page. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/04-add-panel.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/04-add-panel.md new file mode 100644 index 00000000..fed6fb9c --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/04-add-panel.md @@ -0,0 +1,24 @@ +--- +title: Add Chart +permalink: /guide/ee-tenant/dashboard/add-panel/ +--- + +> This document was translated by ChatGPT + +# Add Chart + +A dashboard is composed of one or a group of `charts`. This chapter will explain how to add a `chart` to a `dashboard`. + +Currently, charts can only be added to a `dashboard` from the feature pages, including charts from the `Metrics`, `Application`, and `Network` feature pages. + +**Step 1**: Go to the feature module page where you want to add the chart. For example, click `① Application - Service` to enter the `Service Overview` page. + +![Step 1](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650824ebccb7c.png) + +**Step 2**: Select the chart you want to add to the dashboard, click the `② Settings` button in the chart, and choose the `③ Add to Dashboard` option. + +![Step 2](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650824ed30950.png) + +**Step 3**: Confirm the `④ Name` of the chart and select the `⑤ Dashboard` to which the chart should be added. Click `⑥ Confirm` to successfully add it to the corresponding dashboard. + +![Step 3](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650824edae26e.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/05-variable-template.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/05-variable-template.md new file mode 100644 index 00000000..8135d6ed --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/05-variable-template.md @@ -0,0 +1,114 @@ +--- +title: Template Variables +permalink: /guide/ee-tenant/dashboard/variable-template/ +--- + +> This document was translated by ChatGPT + +# Template Variables + +Template variables allow you to define a set of variables in the current Dashboard, which can be referenced in chart search conditions. By quickly changing the values of these variables, you can update the search conditions of charts without having to create multiple identical visualization Panels just because the search conditions differ. + +## Managing Template Variables + +You can manage template variables centrally through the `Template Variable List`. + +As shown below, in the template variable list, you can `① Add`, `② Delete`, and `③ Edit` variables. You can also enter any string in the `⑤ Search Bar`, and adjust the column width display in `④ Settings`, such as evenly distributing column widths or adjusting them based on content. + +![00-Template Variable List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf59d3ef4d.png) + +## Creating a New Template Variable + +When you need to build quick search conditions for the current Dashboard, you can create a `new template variable` to achieve this. +For example: When building an `Application Observability Dashboard`, if you need to quickly view different `applications`, you can create a template variable for the `application` search condition. + +**Step 1**: On the Dashboard details page, click the `① Settings` button and select `② Manage Template Variables`. + +![01-Step 1](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf68c0b038.png) + +**Step 2**: In the template variable list pop-up, click the `③ New Template Variable` button. + +![02-Step 2](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf70be224b.png) + +**Step 3**: Create the template variable as needed. The DeepFlow platform provides three types of template variables: `Dropdown Selection`, `Text Input`, and `Group`. Detailed descriptions are provided in the following sections. + +![03-Step 3](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf7bb39bbe.png) + +### Dropdown Selection + +A dropdown selection type template variable changes the search condition via a dropdown menu. +Currently, this type supports building template variables for `resource` and `xx_enum` type `Tags` in the DeepFlow platform database. + +- Note: For a description of the DeepFlow platform database, see later sections. + +![04-Dropdown Selection](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024032165fbf81b3b73b.png) + +- **① Query Type:** Different query types pass values differently + - Query by ID: Supports passing IDs for queries + - Query by Name: Supports passing strings for queries +- **② Value Range:** Supports two ways to set template variable values: `Static Values` and `Dynamic Values` + - Static Values: Fixed value range after being referenced + - Dynamic Values: Compared to static values, the range can be affected by `Static Values` or `Value Tag`. For details, see the **Creation and Reference** section +- **③ Data Source:** Specifies the data table where the template variable values come from +- **④ Value Tag:** Specifies the Tag corresponding to the template variable values +- **⑤ Value Range:** Selects the values corresponding to the template variable +- **⑥ Selection Mode:** Changes the search condition via a dropdown menu, single-select by default + - Multi-select: Check `Multi-select` to enable multi-selection mode + - Select All: Check `Select All` to add a `Select All` option in the dropdown, selecting all values for the template variable +- How to reference: When adding a query condition in the chart search bar, enter the Tag. Any existing template variable with the same `Tag` will appear as a dropdown option. For details, see the **Creation and Reference** section. + +#### Creation and Reference + +The following demonstrates how to create and reference `static template variables` and `dynamic template variables`, and how to link dynamic and static template variables. + +- First, create a static template variable named `K8s Namespace` with Tag `pod_ns`, and set its value range to `deepflow-ebpf-istio-demo, deepflow-otel-grpc-demo, deepflow-telegraf-demo`. + +![05-Create Static Template Variable](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd4b0c94b.png) + +- Next, create a dynamic template variable named `K8s Workload` with Tag `pod_group`, and set its value range to `pod_ns = K8s Namespace`. This means the dropdown options for `pod_group` will change based on the selection of `pod_ns`. + +![06-Create Dynamic Template Variable](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd4c9cee2.png) + +- Then, select the query condition `pod_group = K8s Workload` to reference the template variable in the chart search conditions. + +![07-Template Variable Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd4e0596a.png) + +- Finally, you can quickly switch the referenced template variables at the top of the Dashboard. Different selections for `K8s Namespace` will result in different options for `K8s Workload`. + +![08-Using Template Variables](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd504e2f4.png) + +![09-Switch Template Variables](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240402660bbd5108331.png) + +### Text Input + +A text input type template variable changes the search condition by entering a string. + +![10-Text Input](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091865082716002b8.png) + +Text input type template variables can be referenced by any `Tag` or `operator` that allows direct input. +In search conditions, they appear similar to `Dropdown Selection` type template variables. + +- ① Tag Reference: Supports int, int_enum, string, ip, and mac types. All these Tag types support all operators. + +![11-Tag Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309186508271402080.png) + +- ② Operator Reference: Supports :, !:, =~, and !~ operators. All these operators support all Tag types. + +![12-Operator Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091865082716add4e.png) + +### Group + +A group type template variable changes the grouping via a dropdown menu. +For example, when drilling down into data from `K8s Cluster` -> `K8s Namespace` -> `K8s Service` -> `K8s Workload` -> `K8s POD`, you can create this type of template variable. + +![13-Group](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230918650827184c5b7.png) + +- Value Range: All `Tags` in the DeepFlow platform database can be used as values for this type of template variable + - ①: Specifies the data table where the template variable values come from + - ②: Selects the values corresponding to the template variable +- Selection Mode: For details, see the description of the Dropdown Selection type template variable + - Note: The main group cannot reference template variables in `Multi-select` or `Select All` mode + +Group type template variables can only be referenced in the grouping section of search conditions. They appear as dropdown options in the grouping menu. + +![14-Group Reference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091865082715de5ff.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/06-right-slide-box.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/06-right-slide-box.md new file mode 100644 index 00000000..ee04a2af --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/06-right-slide-box.md @@ -0,0 +1,40 @@ +--- +title: Edit Right Slide Box +permalink: /guide/ee-tenant/dashboard/right-slide-box/ +--- + +> This document was translated by ChatGPT + +# Edit Right Slide Box + +The right slide box of a dashboard chart allows you to edit tabs as needed. You can customize the addition or removal of `system tabs` and `dashboard tabs`. Edited tabs will be saved in the dashboard and remembered for each chart. + +- System tabs: Predefined tabs by the DeepFlow system. For details, refer to [Tracing - Right Slide Box](/guide/ee-tenant/tracing/right-sliding-box/) +- Dashboard tabs: Dashboards that users customize and add in a `dashboard` + +![Right Slide Box Tabs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664579a8512bb.png) + +- **① Right Slide Box Tabs:** The display area for tabs, which by default shows all `system tabs` +- **② Tab Management:** Allows you to `sort`, `delete`, and `edit` dashboard tabs +- **③ Add Tab:** See the following sections for details + +## Add Tab + +You can add both `system tabs` and `dashboard tabs` + +### Add System Tab + +![Add System Tab](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664579b508cbe.png) + +**① Select Tab:** Quickly add or remove a system tab by clicking the checkbox. + +### Add Dashboard Tab + +![Add Dashboard Tab](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664579aab5031.png) + +- **① Dashboard Name:** Select the dashboard name to be added to the right slide box tabs. Duplicate additions are not allowed. Once successfully added, the dashboard tab will be loaded in the right slide box in a `read-only` mode. +- **Association Conditions:** Set the values of template variables when the dashboard is loaded in the right slide box + - **② Template Variable:** Template variables in the dashboard + - **③ Template Variable Value:** Set the value of the template variable, which can be one of the following: + - Variable default value: When the dashboard loads, it reads the default value set for the template variable in the dashboard + - $Tag: When the dashboard loads, it reads the value of the Tag corresponding to the search condition when entering the right slide box. For example, if the search condition of the clicked data when entering the right slide box carries `protocol = tcp`, then the `Protocol` template variable will be set to `tcp` when the page loads \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/01-overview.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/01-overview.md new file mode 100644 index 00000000..b4cd785c --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/01-overview.md @@ -0,0 +1,22 @@ +--- +title: Overview +permalink: /guide/ee-tenant/dashboard/panel/overview/ +--- + +> This document was translated by ChatGPT + +# Overview + +A chart is the smallest unit visualization component in DeepFlow. It serves as the visualization component for each functional page and can also be added to a Dashboard to create a customized visualization view. Each chart can be operated independently, and users can modify the search criteria, metrics, styles, and other settings of the chart according to their needs. + +DeepFlow supports various types of charts. Next, we will introduce how to use the following charts: + +- [Traffic Topology](./topology/) +- [Distributed Tracing Flame Graph](./flame/) +- [Line Chart](./line/) +- [Bar Chart](./bar/) +- [Pie Chart](./pie/) +- [Histogram](./histogram/) +- [Table](./table/) +- [Overview Chart](./stat/) +- [Text](./text/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/02-topology.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/02-topology.md new file mode 100644 index 00000000..295e349f --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/02-topology.md @@ -0,0 +1,153 @@ +--- +title: Traffic Topology +permalink: /guide/ee-tenant/dashboard/panel/topology/ +--- + +> This document was translated by ChatGPT + +# Traffic Topology + +The traffic topology in DeepFlow can be used to display the dependencies between services or resources, enabling better analysis and troubleshooting, such as identifying performance bottlenecks, single points of failure, or potential dependency access issues. + +## Overview + +![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031465f2d3476b5c0.png) + +The traffic topology consists of `nodes`, `paths`, and several operations: + +- **① Node:** Represents a service or resource, corresponding to the `group` in the search criteria. It can be a container service, cloud server, region, etc. +- **② Path:** Represents the direction between services or resources, from `client` to `server`. +- **Operations:** You can hover over or click on a `node` or `path` + - Hover: Highlight the `node` or `path` and view metric values + - Click: View details of the `node` or `path` in a right-side sliding panel + +### Topology Details + +![01-Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031465f2cfe0cb741.png) + +- **① Switch Query Region:** If there are multiple storage regions for the data, you can quickly switch regions for querying + - Note: If the query criteria are not grouped, the `Switch TOP Data` function is unavailable +- **② Switch Top Data:** Sorts grouped nodes in descending order based on the primary metric value + - Note: If the query criteria are not grouped, the `Switch TOP Data` function is unavailable +- **③ Expand Table:** Click to expand or collapse the table. For details, see the **Expand Table** section +- **④ Modify Metrics:** Allows you to change the displayed metrics. For details, see the **Modify Metrics** section +- **⑤ Settings:** Click to configure the `traffic topology`. For details, see the **Settings** section +- **⑥ Delete:** A `Dashboard` feature. If you do not want to display this `traffic topology` in the dashboard, click delete to remove it +- **⑦ Manual Resource Relationship Mode:** In this mode, you can manually add paths between nodes +- **⑧ Waterfall/Free Topology:** Switch between topology display styles. Free topology is generally used for scenarios with many nodes and complex paths; waterfall topology is used for fewer nodes and simpler paths +- **⑨ Auto Layout:** The system arranges nodes in a tree structure based on path relationships +- **⑩ Random Layout:** The system arranges nodes in a star structure +- **⑪ Save Topology:** A `Dashboard` feature. After modifying the `traffic topology`, you can click `Save Topology` to save changes such as time range, topology position, topology configuration, and variable template values +- **⑫ Legend:** Open the legend to view the meaning of icons and lines. For details, see the **Legend** section + +### Hover TIP + +![02-HoverTIP](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7b8291e1be.png) + +When hovering over a `node`, the related `paths` are automatically highlighted. When hovering over a `path`, the related `nodes` are highlighted. At the same time, a TIP displays the metric values for different `observation points`. + +### Hover Metric Display + +![03-HoverDisplay](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7b826aff77.png) + +Taking hovering over a `path` as an example, here is the TIP content: + +- First line: Description and legend area. Legend explanation: + - Application Functions + - A Application: Metrics obtained via application instrumentation, currently representing `signal source = OTel` data + - S System: Metrics obtained via eBPF, currently representing `signal source = eBPF` data + - E Endpoint Network: Metrics obtained via traffic capture (BPF) from the NIC of the client or server + - M Middle Network: Metrics obtained via traffic capture (BPF) from locations other than the client or server NIC + - Network Functions: All metrics come from traffic capture (BPF) + - D NIC: Data captured from the NIC of the client or server + - K Container Node: Data captured from the container node NIC + - H Host: Data captured from the host NIC + - M Middle Network: Data captured from NICs other than the above + - Corner Mark: Indicates whether the data collection point is on the client or server + - C: Client + - S: Server +- Second line: Name of the hovered `node/path` +- Others: Metric display area + - Displays metric values based on the data location + - The last column shows the difference across all `observation points`, useful for quickly identifying inconsistencies in `sent traffic` + - Example: As shown below, representing endpoint network data from the client + ![04-Icon](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509427e1c1c9.png) + +### Settings + +Click the gear icon to configure the `traffic topology` + +![05-Settings](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7b8295063a.png) + +- **Show Full Name:** Display full node names or abbreviated names +- **Download CSV Data:** Download the data of the `traffic topology` +- **View API:** View the API information used to generate the `traffic topology` + +### Expand Table + +Click the `Expand Table` button to display the metrics of `nodes` and `paths` in the `traffic topology` in a list format, including resource monitoring, path monitoring, and path difference tables. + +![06-Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965091707f4009.png) + +- Resource Monitoring: Displays metrics for all nodes in the `traffic topology` +- Path Monitoring: Displays metrics for all paths in the `traffic topology` +- Path Difference: Displays the differences in metrics for all paths across different `observation points` +- Search: Quickly search and locate table data +- Settings: Configure column width display, such as equal width or content-based width + +### Modify Metrics + +Metrics are an important part of the chart. DeepFlow provides shortcuts to quickly select metrics to display in the dropdown menu. + +![07-ModifyMetrics](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509170932512.png) + +- **① Metric Name:** Select a metric name to display its data in the chart +- **② Set as Primary Metric:** Click the icon to set the metric as the primary metric. When `Switch TOP Data` is used, the chart is sorted by the primary metric value +- **③ Advanced Settings:** Add, delete, or modify metrics. For details, see the **Advanced Settings** section +- **Multi-select/Single-select:** Some charts support multi-select, where the TIP can display multiple metrics; single-select shows only one metric + +### Advanced Settings + +For further metric configuration, click `Modify Metrics -> Advanced Settings` to enter the settings page. As shown below, you can add or delete metrics, add aggregation functions, modify display names, set thresholds, and select metric templates. + +![08-AdvancedSettings](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965091709aebea.png) + +- **① Select Template:** Choose a `metric template` to quickly switch the current metrics in the popup +- **② Clear All:** Clear all metrics in the popup +- **③ Save Template:** Save the current metric configuration as a `metric template` +- **④ Metric Field:** Click the input box to open a dropdown and select metrics by category + - **Aggregation - Primary Operator:** Aggregate metrics using functions such as average, sum, max, min + - **Aggregation - Secondary Operator:** Perform secondary calculations on data from the `primary operator` + - **Name:** Set the display name of the metric + - **Unit:** Set the display unit of the metric + - **Threshold:** Set a threshold for the metric. When exceeded, the `node` or `path` turns red as an alert +- **⑤ Enable/Disable Metric:** Show or hide the corresponding metric +- **⑥ Add Metric:** Add a `④ Metric Field` in the popup + +### Edit + +The topology editing panel consists of three parts: `① Chart`, `② Search Criteria`, and `③ Style & Settings`. + +![09-Edit](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031365f175fb51d9f.png) + +- **① Chart:** The chart is drawn based on `② Search Criteria` and `③ Style & Settings` +- **② Search Criteria:** For usage, see the **[Search](../../query/overview/)** section +- **③ Style & Settings:** Configure chart styles, colors, etc. + - **Style:** Rich options for chart styling + - **Title:** Modify the chart name + - **Chart Style:** + - Show Full Name: Display full or abbreviated node names + - Show Primary Metric: When enabled, displays the primary metric value on paths + - Disable Thumbnail: Enable/disable the thumbnail view + - Step Line Bend Point: Set the position of bend points in step lines + - Area Fill: Fill the area between the line and the axis with color + - Data Stacking: Display multiple data series stacked or tiled + - Stacked: Stack values of multiple series in the same coordinate system to show overall trends and contributions + - Tiled: Display multiple series side-by-side to better show differences and relationships + - **Color:** Configure path and node colors + +### Legend + +Click `Legend` to view the meaning of icons and lines. + +![10-Legend](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509170b4c72e.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/03-flame.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/03-flame.md new file mode 100644 index 00000000..fe612c7a --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/03-flame.md @@ -0,0 +1,139 @@ +--- +title: Distributed Tracing +permalink: /guide/ee-tenant/dashboard/panel/flame/ +--- + +> This document was translated by ChatGPT + +# Distributed Tracing + +DeepFlow uses `distributed tracing` to present all application Spans, system Spans, and network Spans involved in a single call in one flame graph, enabling true cross-team collaboration among business development teams, framework development teams, service mesh operations teams, container operations teams, DBA teams, and cloud operations teams on a single platform. + +## Overview + +On the `distributed tracing` page, initiate a `tracing` operation for a call, which will then be displayed in a right-side sliding panel. This panel shows the call tracing visualization, as illustrated below. + +``` +Note: Flame graphs and topology graphs in distributed tracing currently do not support being added to a Dashboard +``` + +![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466431461b3f38.png) + +The right-side tracing panel is divided into three parts: header information, data visualization, and call information data list. + +- **① Header Information:** Displays basic information about the trace, such as client, server, request start time, duration, request type, request resource, etc. +- **① Data Visualization:** Displays tracing Span data as a flame graph or shows traced services as a topology graph. +- **② Call Information Data List:** Displays related call information. + +### Flame Graph + +![01-Flame Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965095885c540d.png) + +A flame graph consists of multiple `bar-shaped` blocks, each representing a Span. The x-axis represents time, and the y-axis represents call stack depth. Spans are displayed from top to bottom in the order they are called. Details are as follows: + +- **Length:** Along the x-axis, represents the execution time of a Span, with each end corresponding to the start and end times. +- **Service List:** Shows the proportion of latency consumed by each service. Clicking a service will highlight the corresponding Spans in the `flame graph`. + - **Color:** Application Spans and system Spans use a unique color for each service; all network Spans are gray (as they do not belong to any service). +- **Display Information:** Each bar’s `display information` consists of an `icon` + `call information` + `execution time`. + - Icon: Differentiates Span types: + - A: Application Span, collected via the OpenTelemetry protocol, covering business code and framework code. + - S: System Span, collected via eBPF with zero intrusion, covering system calls, application functions (e.g., HTTPS), API Gateway, and service mesh Sidecar. + - N: Network Span, collected from network traffic via BPF, covering container network components such as iptables, ipvs, OvS, and LinuxBridge. + - Call Information: Varies by Span type: + - Application Span and System Span: `Application protocol`, `Request type`, `Request resource` + - Network Span: `Observation point` + - Execution Time: Total time consumed from Span start to end. +- **Operations:** Supports `hover` and `click`. + - Hover: Displays `call information` + `instance information` + `execution time` in a tooltip. + - Instance Information: Application Span shows `service` + `resource instance`; System Span shows `process` + `resource instance`; Network Span shows `network interface` + `resource instance`. + - Execution Time: Shows the total execution time of the Span and its proportion of self-execution time. + - Click: Highlights the Span and its parent Span, and allows viewing detailed information of the clicked Span. +- **Collapse Sidebar:** Allows collapsing the `service list`. + +### Call Topology Graph + +![02-Call Topology Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023091965095886aa8de.png) + +The call topology graph presents data in an orderly, structured way. Data is aggregated by service into nodes, and horizontal/vertical lines between nodes represent parent-child relationships between Spans, showing the request call relationships. Details are as follows: + +- **Node:** Corresponds to a service in the flame graph’s service list, aggregating one or more Spans under the same service into a single node, and showing the total time consumed by that service in the trace. + - Display Information: Square node `display information` consists of `icon` + `call information` + `self time`. + - Icon: Differentiates Span types; see the Flame Graph section for details. + - Self Time: The total time consumed by one or more Spans corresponding to the service. +- **Path:** Draws topology paths corresponding to the `parent Span` to `child Span` relationships in the flame graph. +- **Operations:** Supports `hover` and `click`; see the Flame Graph section for details. + +### Bottom Tabs + +#### Call Details + +Displays detailed information of Spans in the flame graph in a list format. Clicking a Span in the flame graph highlights the corresponding call detail in the list, and vice versa. + +![Call Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643145809589.png) + +#### IO Events + +When clicking a system Span in the flame graph, if the corresponding process has IO read/write events, you can view them. The IO Events tab allows quick viewing of the time consumed by file read/write operations for the Span. + +![IO Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643145f4f784.png) + +**① First Row:** Overlays IO event blocks from all threads below; the more overlap, the darker the color. +**② Thread Row:** Shows IO events for each thread, with each block representing an event. Block length is calculated from the IO event’s start and end times. + +- Tip: Consists of `file name` + `IO event type` + `event duration` + **③ Detailed Information:** Shows details of the IO event. + +#### Flow Logs + +When clicking a network Span in the flame graph, analyzes latency data from flow logs corresponding to the call log’s time range. + +![Flow Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643145d06086.png) + +**① Status Row:** Determines observation point, flow duration, and flow log status. +**② Latency:** Analyzes network-related latencies, including TCP connection latency, TLS connection latency, average data latency, average system latency, and average client wait latency. For latency calculation methods, see the `metrics diagram`. + +#### Span Tracing Source + +When analyzing why a Span exists in the flame graph, you can use the Span tracing source feature. Clicking a Span in the flame graph displays its relationships with other Spans in a list. DeepFlow’s distributed tracing is computed based on a series of IDs, including TraceID, SpanID, ParentSpanID, request X-Request-ID, response X-Request-ID, request Syscall TraceID, response Syscall TraceID, request TCP Seq number, and response TCP Seq number. When IDs are related, Spans can be linked and displayed in the same flame graph. Related IDs are marked in purple in the list. + +![Span Tracing Source](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466431459e1b6e.png) + +**① Clicked Span:** The Span clicked in the flame graph. +**① Related Span:** Spans related to the clicked Span. + +### Quick Understanding of Flame Graphs + +![Flame Graph Example](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d2abc86b21.png) + +A flame graph represents the passage of time from left to right. In the example trace above, a complete business request is processed as follows: + +- (1) The "Client" process initiates an HTTP GET request, which is transmitted through multiple network interfaces to the "Frontend Service". +- (2) The "Frontend Service", to complete the business process, first sends a DNS query to the "DNS Service", which is transmitted over the network to the "DNS Service". +- (3) The "DNS Service" processes the query and returns a DNS response to the "Frontend Service", transmitted over the network. +- (4) The "Frontend Service" then sends an SQL query, transmitted over the network to the "MySQL Service". +- (5) The "MySQL Service" processes the query and returns an SQL response to the "Frontend Service", transmitted over the network. +- (6) The "Frontend Service" then sends an RPC request, transmitted over the network to the "RPC Service". +- (7) The "RPC Service" processes the request and returns an RPC response to the "Frontend Service", transmitted over the network. +- (8) After receiving the RPC response, the "Frontend Service" sends the final HTTP response to the "Client", transmitted over the network to the "Client". + +The difference in length between any two Spans represents the latency introduced between those two points. + +### Flame Graph Analysis Examples + +- **Example 1: Significant Difference Between Network Spans** + +In the figure below, the significant difference between two network Spans indicates noticeable latency in packet transmission between two network interfaces. If the two interfaces belong to the "Client container node" and the "Server container node", the root cause of the slow response is the forwarding network between the container nodes. + +![Slow Call Flame Graph Example 1 - Significant Network Span Difference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d2926ceffc.png) + +- **Example 2: Significant Difference Between System Spans** + +In the figure below, the significant difference between two system Spans in the "Frontend Service" indicates that the root cause of the slow response lies in the "Frontend Service" process handling. + +![Slow Call Flame Graph Example 2 - Significant System Span Difference](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d292867f8b.png) + +- **Example 3: Noticeably Long Terminal System Span** + +In the figure below, the noticeably long Span in the "DNS Service" indicates that the slow response originates from the "DNS Service" process handling. + +![Slow Call Flame Graph Example 3 - Long Terminal System Span](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660d292a82a96.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/04-line.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/04-line.md new file mode 100644 index 00000000..a60fdbd8 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/04-line.md @@ -0,0 +1,93 @@ +--- +title: Line Chart +permalink: /guide/ee-tenant/dashboard/panel/line/ +--- + +> This document was translated by ChatGPT + +# Line Chart + +A line chart displays continuous data that changes over time, making it ideal for viewing data trends within a specific time range. + +In DeepFlow, line charts are divided into two types: regular line charts and TOP N line charts. + +- Regular line chart: Displays all queried data changes over time. +- TOP N line chart: First groups the queried data and selects the TOP N, then displays the data changes over time for the selected TOP N services or resources. + +## Overview + +![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664b0fd1d1069.png) + +- **① Query Area:** Basic chart operations. For usage details, please refer to [Traffic Topology - Modify Metrics](./topology/). +- **② Modify Metrics:** Basic chart operations. For usage details, please refer to [Traffic Topology](./topology/). +- **③ Settings:** Basic chart operations. For usage details, please refer to [Settings]. +- **④ Delete:** Basic chart operations. For usage details, please refer to [Traffic Topology - Overview](./topology/). + +### Settings + +Users can click the `Settings` button to operate on the chart. + +![01-Settings](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240415661cc68097aa9.png) + +- **Edit:** Allows modification of the chart, such as changing search conditions, name, saving to a different dashboard location, or opening the original chart function page. For usage details, please refer to [Edit]. +- **Copy:** A `Dashboard` feature that supports copying charts within a dashboard. +- **Download CSV Data:** Basic chart operation. For usage details, please refer to [Traffic Topology - Settings](./topology/). +- **View API:** Basic chart operation. For usage details, please refer to [Traffic Topology - Settings](./topology/). + +### Edit + +The line chart edit panel consists of three parts: `① Chart`, `② Search Conditions`, and `③ Configuration`. + +![02-Edit.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664ac8cae92be.png) + +- **① Chart:** The chart is drawn based on `② Search Conditions` and `③ Configuration`. +- **② Search Conditions:** For usage details, please refer to [Search](../../query/overview/). +- **③ Configuration:** Supports quick switching of chart types, as well as style and related function configurations. + - **Switch Chart Type:** Quickly switch chart types. Only charts in dashboards support switching. + - **General Configuration:** Rich features for customizing chart styles. + - **Chart Info:** Edit chart name and add descriptions. + - Title: Modify the chart name. + - Description: Add related description information in markdown format, supporting links, images, etc. + - **Metric Settings:** Set aliases, units, and thresholds for metrics added in `② Search Conditions`. + - **Chart Style:** Configure the display style of the line chart. + - Display Form: Choose from `Line`, `Bar`, or `Point` for plotting. + - Drawing Method: Choose from four ways to connect `lines`. + - Node Display: Data points can be displayed as hollow circles, hollow squares, solid circles, or solid squares. + - Line Style: Data lines can be displayed as solid, dashed, or dotted lines. + - Area Fill: Fill the area between the line and the axis with color. + - Show Values: Display corresponding data series values when hovering with the mouse. + - Data Stacking: Display multiple data series in stacked or tiled form. + - Stacked: Stack values of multiple data series in the same coordinate system to show overall trends and individual contributions. + - Tiled: Display values of multiple data series side-by-side in the same coordinate system to better show differences and relationships. + - **Color:** Set colors for the current chart. + - **Legend:** Configure legend display status, position, values, and display form. + - Mode: Display as legend or table. + - Position: Display below or to the right of the chart. + - Display Values: Choose to display `Avg`, `Max`, `Min`, or `Max` values of metrics. + - **Axis Lines:** Configure background lines and axis lines. + - Background Lines: Set to straight lines, grid lines, or turn off background lines. + - Axis Display: Choose to show or hide axis lines. + - **Data Filtering:** Choose to hide null or zero values. + - **Advanced Configuration:** + - **Tip:** Configure how tips are displayed. + - Tip Mode: Choose from `All`, `Single`, or `Hidden`. + - All: Show data for all series. + - Single: Show data only for the highlighted series under the mouse. + - Hidden: Do not display tips. + - Series Name: Show or hide series names. + - **Data Filtering:** + - Fill Method: Three ways to fill `null data`. + - Fill 0: Default fill method. + - Line Chart: Fill missing data with `0`, tip shows `0`. + - Bar Chart: No bar height, tip shows `0`. + - Fill null: No value at the time point. + - Line Chart: Line breaks, tip shows `null`. + - Bar Chart: No bar height, tip shows `null`. + - Fill none: No time point. + - Line Chart: Skip the point and connect to the next, tip shows the previous point's value. + - Bar Chart: No bar height, tip shows the previous point's value. + - Hide Series: Hide series where `all data is 0` or `all data is null`. + - **Data Sorting:** + - Top Sorting: Sort data in ascending/descending order. + - Top N: Combined with `Top Sorting`, return the smallest/largest N data values. + - Options: `Top 5`, `Top 10`, `Top 20`. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/05-bar.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/05-bar.md new file mode 100644 index 00000000..ab195fae --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/05-bar.md @@ -0,0 +1,31 @@ +--- +title: Bar Chart +permalink: /guide/ee-tenant/dashboard/panel/bar/ +--- + +> This document was translated by ChatGPT + +# Bar Chart + +A bar chart represents data for different categories by drawing a series of vertical or horizontal bars, providing an intuitive view of the size and distribution of the data. In DeepFlow, you can visualize data in the form of a `bar chart` by using the `switch display mode` option in a `table`. + +## Overview + +![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f8001c6c54e.png) + +- **① Switch Query Area:** A basic chart operation. For usage details, please refer to the [Traffic Topology - Overview](./topology/) section. +- **② Switch Top Data:** A basic chart operation. For usage details, please refer to the [Traffic Topology - Overview](./topology/) section. +- **③ Modify Metrics:** A basic chart operation. For usage details, please refer to the [Traffic Topology - Modify Metrics](./topology/) section. +- **④ Settings:** A standard chart operation. For usage details, please refer to the [Line Chart - Settings](./line/) section. +- **⑤ Delete:** A capability within a `Dashboard`. For usage details, please refer to the [Traffic Topology - Overview](./topology/) section. + +### Bar Chart + +The bar chart editing panel consists of three parts: `① Chart`, `② Search Criteria`, and `③ Style & Settings`. + +![01-Bar Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f8001b3666e.png) + +- **① Chart:** The chart is drawn based on `② Search Criteria` and `③ Style & Settings`. +- **② Search Criteria:** For usage of search criteria, please refer to the [Search](../../query/overview/) section. +- **③ Style & Settings:** Configure styles and other settings for the chart. For usage details, please refer to the [Line Chart - Edit](./line/) section. + - **Top Sorting:** Supports ascending/descending sorting of data. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/06-pie.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/06-pie.md new file mode 100644 index 00000000..57b01b04 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/06-pie.md @@ -0,0 +1,17 @@ +--- +title: Pie Chart +permalink: /guide/ee-tenant/dashboard/panel/pie/ +--- + +> This document was translated by ChatGPT + +# Pie Chart + +A pie chart represents data for each category by dividing a whole circle into multiple sectors, providing an intuitive view of the relative size and proportion of the data. In DeepFlow, you can visualize data in the form of a `pie chart` by using the `switch display mode` option in a `table`. + +![Pie Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202309196509754fce717.png) + +- **① Switch Top Data:** A basic chart operation. For details, please refer to the [Traffic Topology - Overview](./topology/) section. +- **② Modify Metrics:** A basic chart operation. For details, please refer to the [Traffic Topology - Modify Metrics](./topology/) section. +- **③ Settings:** A standard chart operation. For details, please refer to the [Line Chart - Settings](./line/) section. +- **④ Delete:** A capability within a `Dashboard`. For details, please refer to the [Traffic Topology - Overview](./topology/) section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/07-histogram.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/07-histogram.md new file mode 100644 index 00000000..5ff6aa79 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/07-histogram.md @@ -0,0 +1,17 @@ +--- +title: Histogram +permalink: /guide/ee-tenant/dashboard/panel/histogram/ +--- + +> This document was translated by ChatGPT + +# Histogram + +A histogram represents the distribution of data by dividing it into several continuous intervals (also called "bins" or "buckets") and plotting the frequency (or relative frequency) of the data within each interval. Histograms help visually display the central tendency, dispersion, and shape of the data. + +![Histogram](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230919650975509aeb6.png) + +- **① Query Area:** Supports switching between different query areas. The query area for a histogram only supports `single selection`. +- **② Modify Metric:** A basic chart operation. For details, please refer to [Traffic Topology - Modify Metric](./topology/). +- **③ Settings:** A standard chart operation. For details, please refer to [Traffic Topology - Settings](./topology/). +- **④ Delete:** A capability within a `Dashboard`. For details, please refer to [Traffic Topology - Overview](./topology/). \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/08-table.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/08-table.md new file mode 100644 index 00000000..798c006b --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/08-table.md @@ -0,0 +1,80 @@ +--- +title: Table +permalink: /guide/ee-tenant/dashboard/panel/table/ +--- + +> This document was translated by ChatGPT + +# Table + +Tables are used to display detailed information of structured data. DeepFlow tables are divided into two types: `Aggregate Table` and `Detail Table`. + +## Aggregate Table + +Aggregate tables support querying data from multiple tables of the same type at the same time, for example, `Service Metrics`, `Path Metrics`, or `xx Logs`. + +![01-Aggregate Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031965f8f90497a37.png) + +- **① Query Area:** Basic chart operation. For usage details, please refer to [Traffic Topology - Modify Metrics](./topology/) section. +- **② Modify Metrics:** Basic chart operation. For usage details, please refer to [Traffic Topology - Overview](./topology/) section. + - Long press and drag data to map the data order to the table. +- **③ Settings:** Basic chart operation. For usage details, please refer to [Traffic Topology - Settings](./topology/) section. +- **④ Delete:** A capability in the `Dashboard`. For usage details, please refer to [Traffic Topology - Overview](./topology/) section. + +### Edit + +The edit panel of an aggregate table consists of three parts: `① Chart`, `② Search Criteria`, and `③ Configuration`. + +![02-Edit](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664aff296a339.png) + +- **① Chart:** The chart is drawn based on `② Search Criteria` and `③ Configuration`. +- **② Search Criteria:** For usage of search criteria, please refer to [Search](../../query/overview/) section. +- **③ Configuration:** Supports quick switching of chart types, and configuring chart styles and related functions. + - **Switch Chart Type:** Basic chart function. For usage details, please refer to [Line Chart](./line/) section. + - **Common Configurations:** Rich functions to set chart styles. + - **Chart Info:** Basic chart function. For usage details, please refer to [Line Chart](./line/) section. + - **Color:** Set the basic color for chart text or background. + - Note: Only effective for columns with `Column Settings - Color` enabled. + - **Column Settings:** Supports setting column color, alignment, value display, etc. + - Color: Choose the coloring target for the configured color — none, text, or background. + - Column Alignment: Choose column alignment — left, center, or right. + - Value Mapping: Match specified column values in three ways and replace them with custom text. + - Text: Match by string. + - Range: Match by numeric range. + - Regular Expression: Match by regular expression. + - Note: Value mapping priority is `Text > Range = Regular Expression`. When conditions have the same priority, the one higher in the list takes effect. + - Threshold: Set a numeric range. Data within the range will display specified colors for text/background. + - Unit: Set the unit for the metric. + - Alias: Set an alias for the metric. + - **Advanced Configurations:** + - **Cell:** Supports configuring the table copy function. + - Copy Function: Enable or disable the ability to copy table content. + - Copy Content: Choose the data content to copy. + - Copy Data: Only copy the data content of the current cell, i.e., `value`. + - Positive Filter Condition: Copy content in the format `key: value`. You can paste it into the page search bar, which will quickly recognize it as a `search tag` for querying. + - Negative Filter Condition: Copy content in the format `key!: value`. You can paste it into the page search bar, which will quickly recognize it as a `search tag` for querying. + - For usage details of search tags, please refer to [Service Search Box](../../query/service-search/) section. + - **Table Settings:** Supports setting the table border and header. + +## Detail Table + +Detail tables only support querying a single type of log data, for example, flow logs or call logs. + +![03-Detail Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031965f8f7906c908.png) + +- **① Query Area:** Basic chart operation. For usage details, please refer to [Traffic Topology - Modify Metrics](./topology/) section. +- **② Column Selection:** Supports searching, adding, and deleting column options available in the current table. + - Long press and drag data to map the data order to the table. +- **③ Settings:** Basic chart operation. For usage details, please refer to [Traffic Topology - Settings](./topology/) section. +- **④ Delete:** A capability in the `Dashboard`. For usage details, please refer to [Traffic Topology - Overview](./topology/) section. + +### Edit + +The edit panel of a detail table consists of three parts: `① Chart`, `② Search Criteria`, and `③ Configuration`. + +![04-Edit](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240520664aff2835cce.png) + +- **① Chart:** The chart is drawn based on `② Search Criteria` and `③ Configuration`. +- **② Search Criteria:** For usage of search criteria, please refer to [Search](../../query/overview/) section. + > Note: Detail tables do not support adding multiple query conditions. +- **③ Configuration:** For usage details, please refer to [Aggregate Table - Edit] section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/09-stat.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/09-stat.md new file mode 100644 index 00000000..21a6a80a --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/09-stat.md @@ -0,0 +1,38 @@ +--- +title: Overview Chart +permalink: /guide/ee-tenant/dashboard/panel/stat/ +--- + +> This document was translated by ChatGPT + +# Overview Chart + +The overview chart displays statistical values based on the query conditions. + +## General Introduction + +![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7e22138b9b.png) + +- **① Query Area:** Basic chart operations. For usage details, please refer to [Traffic Topology - General Introduction](./topology/). +- **② Modify Metric:** Basic chart operations. For usage details, please refer to [Traffic Topology - Modify Metric](./topology/). +- **③ Settings:** Basic chart operations. For usage details, please refer to [Settings]. +- **④ Delete:** Basic chart operations. For usage details, please refer to [Traffic Topology - General Introduction](./topology/). + +### Overview Chart + +The overview chart editing panel consists of three parts: `① Chart`, `② Search Conditions`, and `③ Style & Settings`. + +![01-Overview Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024031865f7e4051ebe2.png) + +- **① Chart:** The chart is drawn based on `② Search Conditions` and `③ Style & Settings`. +- **② Search Conditions:** For usage of search conditions, please refer to [Search](../../query/overview/). +- **③ Style & Settings:** Configure the chart’s style, colors, and other settings. + - **Style:** Rich features are available for customizing the chart’s style. + - **Title:** Supports modifying the chart name. + - **Unit:** Supports editing the unit. If no unit is set, the default unit of the metric will be used. + - **Data Precision:** Supports setting the number of decimal places displayed. + - 1 decimal place, 2 decimal places, 3 decimal places, integer, or full precision (display all decimals of the metric). + - **Color:** Supports setting the font, background, and background image colors. + - **Settings:** + - **Background Chart:** Supports displaying a line chart or bar chart of the metric over the selected time range. + - **Period Comparison:** Supports comparing the current metric value with that of one hour ago or one day ago. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/10-text.md b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/10-text.md new file mode 100644 index 00000000..45cea4d9 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/02-dashboard/99-panel/10-text.md @@ -0,0 +1,17 @@ +--- +title: Text +permalink: /guide/ee-tenant/dashboard/panel/text/ +--- + +> This document was translated by ChatGPT + +# Text + +Text is often used for general descriptions and tips in a dashboard. + +## Overview + +![Overview.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466431ce726d72.png) + +- Supports markdown text format +- Supports adding images, hyperlinks, and more \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/03-universal-map/01-overview.md b/translate/translated/06-guide/02-ee-tenant/03-universal-map/01-overview.md new file mode 100644 index 00000000..db3056e0 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/03-universal-map/01-overview.md @@ -0,0 +1,16 @@ +--- +title: Overview +permalink: /guide/ee-tenant/universal-map/overview/ +--- + +> This document was translated by ChatGPT + +# Overview + +The universal map allows users to define each business independently, as well as each service or service group within the business, enabling users to build their own service topology in a more scenario-oriented way. + +The universal map in DeepFlow is divided into three main pages, which will be described in detail below. + +- [Business Definition](./business-def/) +- [Service List](./service-list/) +- [Service Topology](./service-map/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/03-universal-map/02-business-def.md b/translate/translated/06-guide/02-ee-tenant/03-universal-map/02-business-def.md new file mode 100644 index 00000000..8ad989c6 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/03-universal-map/02-business-def.md @@ -0,0 +1,112 @@ +--- +title: Business Definition +permalink: /guide/ee-tenant/universal-map/business-def/ +--- + +> This document was translated by ChatGPT + +# Business Definition + +A business definition includes the business name, the definition of the data table, as well as the definitions of service groups, services, and paths within the business. The following section explains in detail how to define them. + +![00-Terms Explanation](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f64f1d682.jpg) + +A business can consist of multiple `services` and `paths`. `Services` can be fully user-defined and added to a `custom service group`; alternatively, you can define only the service group, and the `services` will be automatically generated within the group — this is referred to as an `auto-grouped service group`. `Paths` are defined by specifying the access relationships between services. The diagram above illustrates the following: + +- Business: Mobile Banking Business +- Services: + - Independent service not in any group: Frontend Load Service + - Added to a custom service group, frontend service group: Operations Frontend Service, Financial Frontend Service + - Services generated by an auto-grouped service group, middle platform service group: Benefits Center Service, Search Center Service, Average Center Service, Payment Center Service +- Paths: + - Custom: Frontend Load Service -> Operations Frontend Service; Frontend Load Service -> Financial Frontend Service; Operations Frontend Service -> Middle Platform Service Group (auto-grouped); Financial Frontend Service -> Middle Platform Service Group (auto-grouped) + - Auto-generated: Access relationships between services within the Middle Platform Service Group (auto-grouped) + +## Business List + +![01-Business List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9c6679b3.png) + +- **① Create Business**: Supports creating a new business. For details, see the **Create Business** section +- **② Name**: Click to enter the `Business Details Page`. For details, see the **Business Details Page** section +- **③ Star**: Click to `star` or `unstar` a business. The list will be sorted by starred status first, then by name in descending alphabetical order +- **④ Service Topology**: Click to go to the `Service Topology` page to view the current business in a waterfall topology format +- **⑤ Service List**: Click to go to the `Service List` page +- **⑥ Edit**: Edit the business +- **⑦ Delete**: Delete the business + +### Create Business + +![02-Create Business](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e3d393f6.png) + +- Name: Required, the business name +- Data Table: Required, the data table from which the business data is sourced + - To view network metrics for the business, select `Network - Path - Metrics Data (xx)`; to view application metrics, select `Application - Path - Metrics Data (xx)` +- Metrics: Depending on the data table, you can use the corresponding metrics + - Up to 10 metrics can be set + +### Business Details Page + +The business details page consists of two main parts: `Basic Information & Actions` and `Details List`. Here you can define the `services`, `service groups`, and `paths` for the current business. + +#### Services + +![03-Services](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e3e6a56d.png) + +- Basic Information & Actions + - Basic Information: Displays the data table, metrics, and the count of services, service groups, and paths for the business + - Click the count to quickly switch the list below to the corresponding `services`, `service groups`, or `paths` + - Actions: + - Edit: Modify the business name, data table, and metrics. For details, see the **Create Business** section + - Star: After starring, the business will be prioritized in the `Business List` page + - Service Topology: Go to the `Service Topology` page to view the topology of the business. For details, see **[Service Topology](./service-map/)** section + - Service List: Go to the `Service List` page to view the topology of the business. For details, see **[Service List](./service-list/)** section +- Service List: Displays information about all services in the current business, with options to edit or delete + - For example, Redis/DNS/Access Client/Stress Test Client are all independent services + +![04-Create Service](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e3fcabc4.png) + +- Name: Required, service names must be unique within the same business +- Icon: The service icon displayed in `Service Topology` and `Service List`. Currently selectable from `Resource - Icon` +- Filter Conditions: Define service data based on filter conditions, supporting `bi-directional` or `uni-directional` queries + - Set client and server separately: If checked, use separate filter condition boxes for client and server for `uni-directional` filtering; if unchecked, use `bi-directional` filtering + - Direction: Define filter conditions for the service in different roles + - Bi-directional: A single filter condition box applies to both client and server roles + - Uni-directional: Two filter condition boxes for `Server` and `Client` + - Server: Filter conditions when acting as a server + - Client: Filter conditions when acting as a client + - For filter condition operations, see the **[Query](../query/overview/)** section + - Group: Defaults to `*` +- Service Group: Supports adding to a `custom type service group`. A `service` can only belong to one `service group`. For the definition of `service groups`, see the following section +- Metric Thresholds: Adjust metric thresholds for each service + - Metrics are collapsed by default; click the expand button to view and edit + - When a metric exceeds the threshold, the corresponding `service` will be highlighted in red + +#### Service Groups + +![05-Service Groups](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e41e9e32.png) + +- Service Group List: Each row represents a group of services in the business. For example, a frontend service group can include APP access services and Web access services. DeepFlow service groups can be user-defined by adding custom services one by one, or they can be formed from a set of automatically identified services. + +![06-Create Service Group](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e4586d8e.png) + +- Name: Required, service group names must be unique within the same business +- Type: Two types — `Auto-grouped` and `Custom` + - Custom: User-defined, allows manually selecting `services` to join + - Auto-grouped: A service group formed from services automatically identified based on `filter conditions` + - Group: Can be grouped by auto_service or [custom auto-grouping tags](../../../features/auto-tagging/custom-tags) + - For `Direction` and `Filter Conditions` instructions, see the **Services** section +- Metric Thresholds + +#### Paths + +![07-Paths](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e4869fe5.png) + +- Path List: Each row represents a path from a `service`/`service group` to another `service`/`service group`. For example, `Client = Service A, Server = Service B` means that `Service A` accesses `Service B` based on the `client` filter conditions for Service A and the `server` filter conditions for Service B. + - Bulk Delete: Select checkboxes to delete multiple items at once + +![08-Create Path](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024040766124e4a3bd2a.png) + +- Name: Required, path name +- Client: Multi-select, options include `All`, `custom service groups`, and `auto-grouped` service groups +- Server: Multi-select, options include `All`, `custom service groups`, and `auto-grouped` service groups +- All: Selects all custom service groups and auto-grouped service groups \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/03-universal-map/03-service-list.md b/translate/translated/06-guide/02-ee-tenant/03-universal-map/03-service-list.md new file mode 100644 index 00000000..27465505 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/03-universal-map/03-service-list.md @@ -0,0 +1,23 @@ +--- +title: Service List +permalink: /guide/ee-tenant/universal-map/service-list/ +--- + +> This document was translated by ChatGPT + +# Service List + +View the metric data of each service in the business in a list format. + +![01-Service List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240407661262d40a411.png) + +- **Business Switch Dropdown**: Quickly switch between businesses, defaults to the first starred business. +- **Modify Metrics**: Supports showing/hiding metrics in the table. + - Set Primary Metric: The primary metric will be displayed before other metrics. +- **Settings**: Supports `View API`, `Service Management`, etc. +- **Table**: + - Name: Service name, with an icon indicating the service type. + - Service Group: The service group to which the service belongs. + - Region: The region to which the service belongs. + - Metrics: Statistics of the service as a server; will be highlighted in red when exceeding the threshold. + - Actions: Double-click a `table row` to open the right-side panel to view detailed information about the service. For details on using the right-side panel, please refer to the section **[Service Topology - Right-Side Panel](./service-map/)**. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/03-universal-map/04-service-map.md b/translate/translated/06-guide/02-ee-tenant/03-universal-map/04-service-map.md new file mode 100644 index 00000000..bfb45ff2 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/03-universal-map/04-service-map.md @@ -0,0 +1,97 @@ +--- +title: Service Topology +permalink: /guide/ee-tenant/universal-map/service-map/ +--- + +> This document was translated by ChatGPT + +# Service Topology + +Displays the `services` defined by the user in `Business Definition` in a waterfall topology format. + +![01-Service Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9bb16811.png) + +- **① Business Switch Dropdown**: Quickly switch between businesses, defaults to the first starred business +- **② Service Management**: Click the button to enter the business details page +- **③ Modify Metrics**: Supports showing/hiding metrics and setting the primary metric +- **④ Save**: Supports saving the `time range`, `service topology position`, and `configuration` of the service topology +- **⑤ Settings**: Supports functions such as `Edit`, `View API`, `Add to Dashboard`, and `Reset` + - Reset: The service topology will be restored to its initial layout +- **⑥ Service Group**: Consists of a `name row` + `box`. For example, in the figure, Client, gcp-microservices-demo, DNS, and Redis are all independent service groups +- **⑦ Path**: Represents a path with actual data from client to server. Hover to view the TIP, click to view path details in the right sliding panel, see later sections for details +- **⑧ Service**: Each block in the topology represents a service, consisting of a `name row` + `metrics`. Hover to view the TIP, click to view service details in the right sliding panel, see later sections for details + - Name Row: ICON represents the service type + - Metrics: Displays metrics according to the priority of [Observation Points](../../../features/universal-map/auto-metrics) (s-xx > local > rest > app). When a metric exceeds the threshold, the name row and corresponding metric will be highlighted in red +- **⑨ Operation Set**: Supports layout, connection, zoom, and other operations on the topology + - Layout: Enter manual layout mode, supports dragging `services`. Click the `Save` button again to exit and save the layout position + - Edit: Enter path editing mode, supports adding or deleting `paths`. Click the `Edit` button again to exit path editing mode + - Add Path: Supports adding connections between `services/service groups of automatic grouping type`, converting the connection into a `path` + - Delete Path: Click the close button on the path to delete the corresponding `path` + - Zoom In/Out: Zoom in or out of the topology + - Mouse Wheel Zoom: When disabled, the topology size can only be controlled via the `Zoom In/Out` buttons + +## Right Sliding Panel + +Clicking on a `service` or `path` will open the right sliding panel to view its detailed information. The right sliding panel consists of the upper `Call Topology` and the lower TABs. + +### Call Topology + +View the client and server of the selected service through the `Call Topology`, or view the selected path. + +![02-Call Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9aa20975.png) + +- **① Switch Grouping**: View the call topology by \*, auto_service, auto_instance, and [Custom Auto Grouping Tags](../../../features/auto-tagging/custom-tags) +- **② Name**: The name of the currently clicked `service` or `path`, corresponding to the object viewed in the lower TAB +- **③ Node**: Refer to [Traffic Topology](../dashboard/panel/topology/) for details +- **④ Path**: Refer to [Traffic Topology](../dashboard/panel/topology/) for details + +### Knowledge Graph + +![03-Knowledge Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f435c6d.png) + +Refer to [Application - Right Sliding Panel - Knowledge Graph](../tracing/right-sliding-box/) for details + +### Application Performance + +Use `Application Performance` to analyze whether there are application-layer anomalies in the selected service or path. + +![04-Application Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f6ac6b5.png) + +The TAB consists of three charts for throughput, latency, and exceptions, along with an endpoint list below. Click a row in the endpoint list to enter the next-level right sliding panel. + +![04-1-Application Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f764e5d.png) +![04-2-Application Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f799476.png) + +The next-level right sliding panel allows viewing the RED metrics and log details of a specific endpoint. If anomalies exist, `Exception Analysis` can be viewed. + +### Network Performance + +Use `Network Performance` to analyze whether there are application-layer anomalies in the selected service or path. + +![05-Network Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f9625df.png) + +The TAB consists of four charts for throughput, latency, exceptions, and performance, along with a service port list below. Click a row in the list to enter the next-level right sliding panel. + +- Service: View data for the service `as a client` or `as a server` separately +- Path: Click each `observation point` in `Topology Analysis` to view data for each `observation point` separately + +![05-1-Network Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3f9c515d.png) +![05-2-Network Performance](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3fd02700.png) + +The next-level right sliding panel allows viewing the RED metrics and log details of a specific service port. If anomalies exist, `Exception Analysis` can be viewed. + +### Infrastructure + +Use `Infrastructure` to analyze CPU, memory, status, and other data of the infrastructure instances corresponding to the service. + +![06-Infrastructure](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a9ae67608.png) + +**① Switch Service:** Switch the service whose infrastructure you want to view. The options are the two services at both ends of the clicked path or the clicked service itself. + +### Events + +View resource change events for a `service` or `path`. + +![06-Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310196530f3fcdb8b4.png) + +Refer to [Tracing - Right Sliding Panel - Events](../tracing/right-sliding-box/) for details \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/01-overview.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/01-overview.md new file mode 100644 index 00000000..b2cbb3d5 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/01-overview.md @@ -0,0 +1,20 @@ +--- +title: Overview +permalink: /guide/ee-tenant/tracing/overview/ +--- + +> This document was translated by ChatGPT + +# Overview + +The DeepFlow application module supports real-time monitoring of service golden metrics, presenting service call topology, performing in-depth analysis of service call logs, and initiating blind-spot-free distributed tracing. By adopting a zero-intrusion approach to applications, it enables application services to achieve observability, allowing users to efficiently identify and locate performance bottlenecks at the application layer, quickly detect and resolve errors and anomalies in application services, and carry out targeted optimizations. This significantly improves the performance and reliability of application services. + +The DeepFlow application consists of six main pages, which will be described in detail below. + +- [Resource Analysis](./service-list/) +- [Path Analysis](./service-statistics/) +- [Topology Analysis](./path-topology/) +- [Call Logs](./call-log/) +- [Distributed Tracing](./call-chain-tracing/) +- [File Reading and Writing](./file-reading-and-writing/) +- [Right Sliding Box](./right-sliding-box/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/02-service-list.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/02-service-list.md new file mode 100644 index 00000000..ad10f100 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/02-service-list.md @@ -0,0 +1,32 @@ +--- +title: Resource Analysis +permalink: /guide/ee-tenant/tracing/service-list/ +--- + +> This document was translated by ChatGPT + +# Resource Analysis + +The Resource Analysis page provides a centralized way to present an overview of the application services monitored by DeepFlow. It includes information such as the name, source, and golden metrics of each application service. Through the Resource Analysis page, users can quickly obtain the overall status of application services, easily locate the services that require attention, and further view detailed information for performance analysis, troubleshooting, and optimization. This helps improve the efficiency and accuracy of service monitoring. + +## Overview + +The Resource Analysis page supports application service overview queries through time filtering and conditional search. + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a602e67679.png) + +- **① Time Selector**: Supports time filtering queries. For usage details, please refer to [Dashboard Details - Time Selector](../dashboard/use/). +- **② Search Snapshot**: Supports saving search conditions as snapshots. For usage details, please refer to [Query - Search Snapshot](../query/history/). +- **③ Search Save and Settings**: + - Search Save: Supports quickly saving the current page's search conditions. For usage details, please refer to [Query - Search Snapshot](../query/history/). + - Settings: A collection of page setting operations + - Database Fields: Supports viewing the tags and metrics in the data table used on the current page + - Enable/Disable Tip Sync: When enabled, you can view metric data for all line charts at the same time point + - Switch Fill Method: When data is missing at a certain time point, you can switch the fill method as needed + - Toggle Stacking: Quickly switch between tiled/stacked display for all time-series related Panels on the page + - Name Abbreviation/Full Display: Toggle between displaying the full name or abbreviated name in the legend +- **④ Search Box**: Supports searching or grouping by Tag. For usage details, please refer to [Query](../query/overview/). +- **⑤ Left Quick Filter**: Allows quick data filtering. For usage details, please refer to [Query - Left Quick Filter](../query/left-quick-filter/). +- **⑥ Region Query**: Supports quickly switching query data by region +- **Table Operations**: + - Click Row: Clicking on a table row allows you to quickly open the right sliding box to view related information of the corresponding application service. For usage details, please refer to [Right Sliding Box](./right-sliding-box/). \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/03-service-statistics.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/03-service-statistics.md new file mode 100644 index 00000000..51a7f811 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/03-service-statistics.md @@ -0,0 +1,30 @@ +--- +title: Path Analysis +permalink: /guide/ee-tenant/tracing/service-statistics/ +--- + +> This document was translated by ChatGPT + +# Path Analysis + +The Path Analysis page, based on the Resource Analysis page, displays both the client and server of application requests. It allows for more flexible and multi-dimensional analysis of application performance metrics, providing insights into service request rate, response time, and error ratio, which helps identify system bottlenecks and optimize performance. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a6ba9e6900.png) + +- **① Time Selector**: Supports time-based filtering queries. For details, please refer to [Dashboard Details - Time Selector](../dashboard/use/). +- **② Search Snapshot**: Supports saving search conditions as snapshots. For details, please refer to [Query - Search Snapshot](../query/history/). +- **③ Search Save and Settings**: + - Save Search: Supports quickly saving the current page's search conditions. For details, please refer to [Query - Search Snapshot](../query/history/). + - Settings: A collection of page setting operations. + - Database Fields: Supports viewing the tags and metrics used in the data table on the current page. + - Enable/Disable Tip Sync: When enabled, you can view metric data for all line charts at the same time point simultaneously. + - Switch Interpolation Method: When data is missing at a certain time point, you can switch the interpolation method as needed. + - Toggle Stacking: Quickly switch between tiled/stacked display formats for all time-series related Panels on the page. + - Name Abbreviation/Full Name Display: Choose whether to display legend names in full or abbreviated form. +- **④ Search Box**: Supports searching or grouping by Tag. For details, please refer to [Query](../query/overview/). +- **⑤ Left Quick Filter**: Allows quick data filtering. For details, please refer to [Query - Left Quick Filter](../query/left-quick-filter/). +- **⑥ Region Query**: Supports quickly switching query data by region. +- **Table Operations**: + - Click Row: Clicking on a table row allows you to quickly open the right sliding box to view related information about the corresponding application service. For details, please refer to [Right Sliding Box](./right-sliding-box/). \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/04-path-topology.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/04-path-topology.md new file mode 100644 index 00000000..e7945db8 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/04-path-topology.md @@ -0,0 +1,26 @@ +--- +title: Topology Analysis +permalink: /guide/ee-tenant/tracing/path-topology/ +--- + +> This document was translated by ChatGPT + +# Topology Analysis + +The topology analysis page displays the dependencies between services or resources in the form of a topology. Combined with threshold values for metrics, it enables quick identification of bottlenecks and issues in the system, allowing timely actions for response and resolution. In addition, by continuously monitoring and updating the topology analysis paths, the system architecture and performance can be continuously optimized. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650a6d7f5039d.png) + +- **① Time Selector**: Supports time-based filtering queries. For usage details, please refer to [Dashboard Details - Time Selector](../dashboard/use/). +- **② Search Snapshot**: Supports saving search conditions as snapshots. For usage details, please refer to [Query - Search Snapshot](../query/history/). +- **③ Search Save and Settings**: + - Save Search: Supports quickly saving the current page's search conditions. For usage details, please refer to [Query - Search Snapshot](../query/history/). + - Settings: A collection of page setting operations. + - Database Fields: Supports viewing the tags and metrics from the data table used on the current page. + - Name Abbreviation/Full Name Display: Choose to display the legend names in full or abbreviated form. +- **④ Search Box**: Supports searching or grouping by Tag. For usage details, please refer to [Query](../query/overview/). +- **⑤ Left Quick Filter**: Allows quick filtering of data. For usage details, please refer to [Query - Left Quick Filter](../query/left-quick-filter/). +- **⑥ Region Query**: Supports quickly switching the query region data. +- **Topology Graph Operations**: Double-click a data node to open the right-side panel and view the corresponding information. For usage details, please refer to [Traffic Topology](../dashboard/panel/topology/). \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/05-call-log.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/05-call-log.md new file mode 100644 index 00000000..034e2ef4 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/05-call-log.md @@ -0,0 +1,28 @@ +--- +title: Call Log +permalink: /guide/ee-tenant/tracing/call-log/ +--- + +> This document was translated by ChatGPT + +# Call Log + +The call log records detailed information for each call. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cbbf9e6ac2.png) + +- **① Time Selector**: Supports time-based filtering queries. For usage details, please refer to [Dashboard Details - Time Selector](../dashboard/use/). +- **② Search Snapshot**: Supports saving search criteria as snapshots. For usage details, please refer to [Query - Search Snapshot](../query/history/). +- **③ Search Save and Settings**: + - Save Search: Supports quickly saving the current page's search criteria. For usage details, please refer to [Query - Search Snapshot](../query/history/). + - Settings: A collection of page configuration operations. + - Database Fields: Supports viewing the tags and metrics from the data table used on the current page. + - Name Abbreviation/Full Display: Choose whether to display the full name or abbreviated name in the legend. +- **④ Service Search Box**: Supports searching or grouping by Tag. For usage details, please refer to [Query](../query/overview/). +- **⑤ Left Quick Filter**: Allows quick data filtering. For usage details, please refer to [Query - Left Quick Filter](../query/left-quick-filter/). +- **⑥ Region Query**: Supports quickly switching the query to different regions. +- **Table Operations**: + - Click Row: Clicking a row in the table allows you to quickly open the right sliding panel to view related information about the corresponding application service. For usage details, please refer to [Right Sliding Box](./right-sliding-box/). + - ⑦ Copy: Table content copy function. After clicking, you can choose the format of the content to copy. For configuration details, please refer to [Table](../dashboard/panel/table/). \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/06-call-chain-tracing.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/06-call-chain-tracing.md new file mode 100644 index 00000000..76854f42 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/06-call-chain-tracing.md @@ -0,0 +1,27 @@ +--- +title: Distributed Tracing +permalink: /guide/ee-tenant/tracing/call-chain-tracing/ +--- + +> This document was translated by ChatGPT + +# Distributed Tracing + +Distributed tracing records detailed information for each call, and only supports traces initiated from data collected via eBPF or transmitted to DeepFlow through the OpenTelemetry protocol. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566442db5b7387.png) + +- **① Time Selector**: Supports time-based filtering queries. For details, please refer to [Dashboard Details - Time Selector](../dashboard/use/). +- **② Search Snapshot**: Supports saving search conditions as snapshots. For details, please refer to [Query - Search Snapshot](../query/history/). +- **③ Search Save and Settings**: + - Save Search: Supports quickly saving the current page's search conditions. For details, please refer to [Query - Search Snapshot](../query/history/). + - Settings: A collection of page setting operations. + - Database Fields: Supports viewing tags and metrics from the data table used on the current page. + - Name Full/Abbreviated Display: Choose to display legend names in full or abbreviated form. +- **④ Service Search Box**: Supports searching or grouping by Tag. For details, please refer to [Query](../query/overview/). +- **⑤ Left Quick Filter**: Allows quick data filtering. For details, please refer to [Query - Left Quick Filter](../query/left-quick-filter/). +- **⑥ Region Query**: Supports quickly switching the query region data. +- **Application Tracing Table**: Displays call information between services or resources within a certain time range, such as client, server, requested resource, request type, request domain name, etc. For details, please refer to [Table](../dashboard/panel/table/). + - **Action:** Click a table row to open the right sliding panel and view the entire call lifecycle traced from that request. For details, please refer to [Right Sliding Panel - Distributed Tracing](./right-sliding-box/). \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/07-file-reading-and-writing.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/07-file-reading-and-writing.md new file mode 100644 index 00000000..eec4aea1 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/07-file-reading-and-writing.md @@ -0,0 +1,14 @@ +--- +title: File Reading and Writing +permalink: /guide/ee-tenant/tracing/file-reading-and-writing/ +--- + +> This document was translated by ChatGPT + +# File Reading and Writing + +The File Reading and Writing page records file read and write operations. Users can view key information such as the time of the operation, the user performing the read/write, and the file path, enabling timely detection and handling of abnormal behavior. + +![3_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650becce082cd.png) + +- You can use the `Column Options` feature in the table to add the event information you want to view. For detailed usage, please refer to the **[Tracing - Call Log](../tracing/call-log/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/04-tracing/08-right-sliding-box.md b/translate/translated/06-guide/02-ee-tenant/04-tracing/08-right-sliding-box.md new file mode 100644 index 00000000..65bdf505 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/04-tracing/08-right-sliding-box.md @@ -0,0 +1,210 @@ +--- +title: Right Sliding Panel +permalink: /guide/ee-tenant/tracing/right-sliding-box/ +--- + +> This document was translated by ChatGPT + +# Right Sliding Panel + +Clicking on a table row, line chart legend, topology diagram, or similar elements on a functional page will bring up the right sliding panel, which displays detailed information about the clicked data. The right sliding panel offers multiple features, including Knowledge Graph, Traffic Relationship, Application Metrics, Endpoint List, Call Logs, Distributed Tracing, Network Metrics, Network Path, Flow Logs, NAT Tracing, Events, and more, to meet different user needs. You can select the corresponding feature based on actual requirements to view and analyze data, enabling faster problem detection and resolution, and improving work efficiency. + +The following sections provide detailed instructions for each feature. + +## Knowledge Graph + +The Knowledge Graph displays all Tags associated with the clicked data object in both list and topology formats. + +![Knowledge Graph](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab0460298a.png) + +- **Knowledge List**: Displays all Tags associated with the clicked data object in key-value pairs, categorized into `Universal Tag (resource tags)`, `Custom Tag`, and `others`. + - Note: If the clicked data is from the `Call Series Page`, categories will be distinguished between Client and Server. + - Operations: Supports searching, category filtering, and empty value filtering for key-value pairs. + - Search: Supports quick search across data in `Select All`. + - Left-side category filter: Check the categories to display for quick content filtering. + - Show/Hide empty tags: Show or hide tags with a value of `--`. + - Hover over a tag and click the `copy` icon to quickly copy it. + - Paste the copied content into the search bar, which can be converted into a search tag. For details on using search tags, refer to the **[Service Search Box](../query/service-search/)** section. +- **Knowledge Graph**: Displays the relationships between Tags in a star topology. Clicking a node shows its associated nodes. + +## Traffic Relationship + +The upper section displays upstream and downstream metrics of the clicked data object in a table. Clicking a row uses DeepFlow’s proprietary flow tracing algorithm to trace the `access data` through observation points in the virtual or physical network. + +![Traffic Relationship](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fad66aac.png) + +- **① Dropdown:** Click to select the object whose traffic relationship you want to view. If the clicked data is from the `Path Data Page`, there will be two objects. +- **② Role:** Check `As Client` or `As Server` to view metrics when the current object acts as a `Client` or `Server`. + +![Virtual - Link Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fae1d9bb.png) + +The link topology is sorted from left to right according to the observation points the `access data` passes through, e.g., Client Application -> Client Process -> Client NIC -> ... -> Server NIC -> Server Process -> Server Application. + +- Note: Each `node` in the topology represents aggregated information from the same observation point. +- Hover: View metric information. + +![Virtual - Detail Table](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fafcb207.png) + +The detail table shows detailed information for each observation point of the `access data`, including the resource it belongs to, data collection location, tunnel information, and metric details. + +- Clicking a row will open the detailed information of the clicked observation point in the right sliding panel. + +![Virtual - Bar Chart](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240515664460986c6da.png) + +The bar chart displays the same data as the link topology but in a bar format for easier visual comparison. + +- **① Latency Difference:** For example, the chart above shows the response latency metrics of access data at each observation point. The differences between bars clearly indicate a significant latency bottleneck from the `Client Container Node` to the `Server Container Node` compared to other points. + +![Physical Topology](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fb6e1309.png) + +When `access data` passes through the `physical network`, it is displayed in a physical topology showing the metric values collected at each physical collection point. + +- `Node` data is obtained from the corresponding `network location`. If no data is collected at a location for the current `access data`, it will be shown as empty. + +## Application Metrics + +Application metrics display aggregated values of application metrics over a period in an overview chart. Multiple line charts can be added to show metric trends over time. + +![Application Metrics](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab046aa9e5.png) + +- **Metric Curve** + - Metric Name: Click to select the metric line chart to display. + - Aggregation Function: Supports applying functions to the selected metric. + - Grouping: Supports grouping the current data. For example, when viewing the response latency of a service, you can further group by `l7_protocol` to see latency per application protocol. + - Enable/Disable Tip Sync: When enabled, view metric values at the same time point across all line charts. +- Click the time component in the upper right to filter data by time. + +## Endpoint List + +The endpoint list displays metrics grouped by `endpoint`. + +![Endpoint List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445f9f99c90.png) + +- Click a table row to enter the `Call Logs` page for that `endpoint`. For details, see the **Call Logs** section. +- Click the time component in the upper right to filter data by time. + +## Call Logs + +Call logs display detailed call log data for the clicked item in both trend analysis chart and table formats. + +![Call Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab2c948e98.png) + +- Trend Analysis Chart: Shows log collection over a time range. Users can select any time segment to zoom in and view logs for that period. +- Log Detail Table: Displays log information such as client, server, application protocol, request type, request domain, etc., and dynamically updates based on the trend chart. For details, see **[Table](../dashboard/panel/table/)**. + - Click a table row to enter the log detail page. For details, see **Call Log Details**. +- Click the icon in the upper right of the trend chart to open the `Call Logs` page in a new tab for custom searches. +- Click the time component in the upper right to filter data by time. + +## Call Log Details + +The call log details page shows response latency, application protocol, request type, request resource, and response status between the client and server at the top. Below the basic information are two buttons with different actions depending on the log source. Below the buttons are the Tags and Metrics for the log, displayed as key-value pairs for quick reference. + +![Call Log Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab59c80b81.png) + +- Action Buttons: + - View Flow Logs: Available only for logs with `Packet` as the signal source. For details, see **Flow Log Details**. + - Distributed Tracing: Available for logs with `eBPF` or `OTel` as the signal source. For details, see **[Distributed Tracing](../dashboard/panel/flame/)**. +- For tag search, filtering, and other `operations`, see **Knowledge Graph**. + +## Network Metrics + +Network metrics display aggregated values over a period and show trends in a line chart. + +![Network Metrics](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab47494a85.png) + +- Metric Name: Click to select the metric line chart to display. +- Aggregation Function: Supports applying functions to the selected metric. +- Grouping: Supports grouping the current data. For example, when viewing the traffic size of a cloud server, you can further group by `server_port` to see traffic per port. +- Enable/Disable Tip Sync: When enabled, view metric values at the same time point. +- For details on using line charts, see **[Line Chart](../dashboard/panel/line/)**. +- Click the time component in the upper right to filter data by time. + +## Flow Logs + +Flow logs record detailed information for each flow at a one-minute granularity, displayed in both trend analysis chart and table formats. + +![Flow Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cbef482a14.png) + +- Trend Analysis Chart: Shows flow log collection over a time range. Users can select any time segment to zoom in and view logs for that period. +- Log Detail Table: Displays flow log information such as client, server, application protocol, request type, request domain, etc., and dynamically updates based on the trend chart. For details, see **[Table](../dashboard/panel/table/)**. + - Left Quick Filter: Filter flow logs using the left sidebar. For details, see **[Left Quick Filter](../query/left-quick-filter/)**. + - Click a table row to enter the log detail page. For details, see **Flow Log Details**. +- Click the icon in the upper right of the trend chart to open the `Flow Logs` page in a new tab. +- Click the time component in the upper right to filter data by time. + +## Flow Log Details + +Flow log details provide further information. For `TCP` protocol logs, TCP sequence diagram tracing and NAT tracing are available. + +![Flow Log Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cf395cfbd9.png) + +- Action Buttons: + - TCP Sequence Diagram: Displays detailed information for each TCP packet header, showing the process of connection establishment, data transfer, and connection closure. For details, see **TCP Sequence Diagram Analysis**. + - NAT Tracing: Initiates tracing using a `five-tuple`. For details, see **NAT Tracing Details**. + - PCAP Download: If a matching PCAP policy exists, download is available. + - Call Logs: Displays current call log information (see below **14 - Call Logs**). +- For `operations`, see **Knowledge Graph**. + +![Call Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240403660cf4122132c.png) + +- Consists of a call trend analysis chart and call log table. + - Trend Analysis Chart: Shows the number of requests over a time range. + - Call Log Table: Displays detailed call log information. Click a row to enter the `Call Log Details` page. + +## TCP Sequence Diagram + +The TCP sequence diagram displays detailed information for each TCP packet header in both trend analysis chart and table formats, including timestamps, direction, flags, Seq, Ack, etc., during connection establishment, data transfer, and connection closure, for deeper network communication analysis and troubleshooting. + +![17-TCP Sequence Diagram](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650aba4f61574.png) + +- Click the time component in the upper right to filter data by time. +- Sequence Table: + - Time, Seq, and Ack can be toggled between relative and absolute values via the list header buttons. + - Interval time is the difference between the current and previous row. + - PCAP Download: If the current flow matches a PCAP policy, the PCAP file can be downloaded. + +## NAT Tracing + +NAT tracing can initiate tracing for any TCP four-tuple or five-tuple, using DeepFlow’s proprietary algorithm to automatically trace traffic before and after NAT. The NAT tracing page displays metrics for the four-tuple in a table. Clicking `Trace` initiates tracing. + +![NAT Tracing](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ab59bad70a.png) + +- Click a table row to trace the data. For details, see **NAT Tracing Details**. +- Click the icon in the upper right to open the `NAT Tracing` page in a new tab for custom searches. +- Click the time component in the upper right to filter data by time. + +## NAT Tracing Details + +NAT tracing details are divided into three parts: the header with information about the traced flow, the left side showing the traced network topology (virtual and physical), and the right side showing the corresponding traffic topology. + +![NAT Tracing Details](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650aba4daabc0.png) + +- Traffic Topology: Displays all traced traffic in a `free topology`. Click a `line` to view detailed path information. +- Network Topology: Displays observation points of the traced traffic in a `waterfall topology`. + - Virtual Network Topology: Shows observation points in the virtual network, such as client NIC, client container node, server container node, server NIC. + - Physical Network Topology: Shows network locations in the physical network. + - Note: Only displayed if the traced traffic passes through the physical network. + - Operations: + - Hover over `nodes` and `lines` to view detailed information, including observation point, network location, tunnel info, and metrics. +- Action Buttons: + - Change Metrics: Switch the queried metrics. + - Show/Hide Delta: Show or hide the difference in primary metrics between adjacent nodes in the `network topology`. + - Show/Hide Full Names: Show or hide full node names in the topology. + - Settings: + - Add to Dashboard: Add the Panel to a Dashboard. + - View API: Basic Panel operation. See **[Traffic Topology - Settings](../dashboard/panel/topology/)**. + - Toggle Thumbnail: Show/hide the thumbnail in the lower left of the topology. + - Matching Algorithm Degree: Adjust NAT tracing algorithm parameters. +- Click the time component in the upper left to filter data by time. + +## Resource Change Events + +Displays resource change events for the clicked data object in both trend analysis chart and table formats. + +![Resource Change Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fa11ff96.png) + +## File Read/Write Events + +Displays file read/write events for the clicked data object in both trend analysis chart and table formats. + +![File Read/Write Events](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566445fa2b14d2.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/05-profiling/01-continue-profile.md b/translate/translated/06-guide/02-ee-tenant/05-profiling/01-continue-profile.md new file mode 100644 index 00000000..6e429125 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/05-profiling/01-continue-profile.md @@ -0,0 +1,40 @@ +--- +title: Continuous Profiling +permalink: /guide/ee-tenant/profiling/continue-profile/ +--- + +> This document was translated by ChatGPT + +# Continuous Profiling + +For the working principle, see [Core Features - Continuous Profiling Description](../../../features/continuous-profiling/auto-profiling) + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146642dfb068b35.png) + +- **① Page Search**: Search bar, search snapshots, and other functions. For usage details, please refer to the section [Tracing - Resource Analysis](../tracing/service-list/) +- **② Quick Filter**: Supports filtering by `Application List` and `Profiling Type` + - Application List: Displays the list of reported services. The default display format is **Service Name (Language Type)** + - Profiling Type: Displays the `Profiling Types` supported by the profiling method configured on the client + - Select the `Profiling Type` supported by the current application. Different profiling types represent different semantics +- **③ Display Switch**: Switch the display mode of performance profiling data. Currently supports: Flame Graph, Table, and Both + - Flame Graph: Displays the function call stack in the form of a flame graph + - **⑥ Tip**: Hover to view information about the Span in the flame graph + - Function Type: + - K: Linux kernel function + - L: Function in a dynamic link library + - A: Application business function + - P: Process + - T: Thread, only appears on the second layer of the flame graph + - ?: Unknown, function name failed to translate. For detailed explanation, see [Core Features - Continuous Profiling - Viewing Data - About Function Type](../../../features/continuous-profiling/data/) + - Span Name + - Total Consumption: Percentage of the Span's total consumption relative to the root (first row of the flame graph) + - Self Consumption: Percentage of the Span's self consumption relative to the root (first row of the flame graph) + - Operation: Click to zoom in and view the call stack of the clicked Span; click on a blank area to return to the original state + - Table: Displays `Self Consumption` and `Total Consumption` of functions in a list format + - Default sorting is in descending order by `Self Consumption` + - Both: View performance profiling data in both `Flame Graph` and `Table` formats simultaneously + - Clicking a function in the table will highlight it in the flame graph +- **④ Flame Graph Name Display**: Choose whether to display the head or tail of the name in the flame graph +- **⑤ Data Filter**: Enter characters to filter data in the table \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/01-overview.md b/translate/translated/06-guide/02-ee-tenant/06-network/01-overview.md new file mode 100644 index 00000000..4479bcb6 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/01-overview.md @@ -0,0 +1,21 @@ +--- +title: Overview +permalink: /guide/ee-tenant/network/overview/ +--- + +> This document was translated by ChatGPT + +# Overview + +The DeepFlow network module offers a wide range of features to help users monitor path traffic and network performance in real time, including resource node information, flow log data, NAT tracing, PCAP download, traffic distribution, and resource inventory. It enables real-time monitoring, analysis, and evaluation of performance metrics such as transmitted traffic, latency, and packet loss rate, allowing timely detection and resolution of network congestion, failures, and security issues. This improves network stability and reliability, ensuring efficient network operation. + +The following sections provide detailed instructions and explanations for each page. + +- [Resource Analysis](./service-statistics/) +- [Path Analysis](./network-path/) +- [Topology Analysis](./network-map/) +- [Flow Logs](./flow-log/) +- [NAT Tracing](./NAT-traversal/) +- [Resource Inventory](./resource-inventory/) +- [PCAP Policy](./pacp-strategy/) +- [PCAP Download](./pcap-download/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/02-service-statistics.md b/translate/translated/06-guide/02-ee-tenant/06-network/02-service-statistics.md new file mode 100644 index 00000000..be4032cf --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/02-service-statistics.md @@ -0,0 +1,16 @@ +--- +title: Resource Analysis +permalink: /guide/ee-tenant/network/service-statistics/ +--- + +> This document was translated by ChatGPT + +# Resource Analysis + +Resource Analysis displays network traffic-related information at various network locations of network flows in the form of line charts and lists. Users can quickly obtain network traffic conditions and the real-time status of each network node. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650c09d0c6aea.png) + +- For details on the functions of the page buttons, please refer to the **[Tracing - Resource Analysis](../tracing/service-list/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/03-network-path.md b/translate/translated/06-guide/02-ee-tenant/06-network/03-network-path.md new file mode 100644 index 00000000..9e004e0d --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/03-network-path.md @@ -0,0 +1,16 @@ +--- +title: Path Analysis +permalink: /guide/ee-tenant/network/network-path/ +--- + +> This document was translated by ChatGPT + +# Path Analysis + +Path analysis presents traffic information collected at network locations in the data link in the form of line charts and tables. Using DeepFlow's self-developed flow tracing algorithm, it can trace the observation points that network flows or application calls pass through in virtual or physical networks. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac4cf837e7.png) + +- For details on the page button functions, please refer to the **[Tracing - Path Analysis](../tracing/service-statistics/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/04-network-map.md b/translate/translated/06-guide/02-ee-tenant/06-network/04-network-map.md new file mode 100644 index 00000000..a8ebf0df --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/04-network-map.md @@ -0,0 +1,16 @@ +--- +title: Topology Analysis +permalink: /guide/ee-tenant/network/network-map/ +--- + +> This document was translated by ChatGPT + +# Topology Analysis + +The topology analysis page displays the relationships between observation points of network traffic within virtual or physical networks in the form of a topology. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac4d081034.png) + +- For details on the page button functions, please refer to the **[Tracing - Topology Analysis](../tracing/path-topology/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/05-flow-log.md b/translate/translated/06-guide/02-ee-tenant/06-network/05-flow-log.md new file mode 100644 index 00000000..cbddedaf --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/05-flow-log.md @@ -0,0 +1,16 @@ +--- +title: Flow Log +permalink: /guide/ee-tenant/network/flow-log/ +--- + +> This document was translated by ChatGPT + +# Flow Log + +Flow logs record detailed information for each flow at a per-minute granularity, and present the flow log data through trend analysis charts and tables. Data collected from network locations along the link is parsed layer by layer, then organized after AutoTagging to produce the flow logs. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac4d20944c.png) + +- For details on the page button functions, please refer to the **[Tracing - Call Log](../tracing/call-log/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/06-NAT-traversal.md b/translate/translated/06-guide/02-ee-tenant/06-network/06-NAT-traversal.md new file mode 100644 index 00000000..ed26a4ed --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/06-NAT-traversal.md @@ -0,0 +1,19 @@ +--- +title: NAT Tracing +permalink: /guide/ee-tenant/network/NAT-traversal/ +--- + +> This document was translated by ChatGPT + +# NAT Tracing + +NAT tracing can initiate tracing for any TCP four-tuple or five-tuple, and uses DeepFlow’s proprietary algorithm to automatically trace traffic before and after NAT. The NAT tracing page displays the metrics corresponding to the four-tuple of the clicked data in a table format. Clicking `Trace` will initiate tracing for the four-tuple. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566442f3ee84e5.png) + +- For the functions of the page buttons and detailed usage, please refer to the **[Tracing - Call Log](../tracing/call-log/)** section. +- Network tracing table: Displays client, server, group information, traffic rate, TCP retransmission ratio, TCP connection failures, and TCP connection latency in a table format. + - Operations: + - Click a row: You can trace the data. For detailed usage, please refer to the **[Tracing - Right Sliding Panel - NAT Tracing Details](../tracing/right-sliding-box/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/07-resource-inventory.md b/translate/translated/06-guide/02-ee-tenant/06-network/07-resource-inventory.md new file mode 100644 index 00000000..3e0b05c7 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/07-resource-inventory.md @@ -0,0 +1,16 @@ +--- +title: Resource Inventory +permalink: /guide/ee-tenant/network/resource-inventory/ +--- + +> This document was translated by ChatGPT + +# Resource Inventory + +Resource inventory involves taking stock of and managing various resources required by applications in the system, including computing resources, network resources, storage resources, and more. Typically, these resources are provided to applications in a virtualized manner. Through resource inventory, system administrators can gain a better understanding of resource allocation and utilization, enabling more efficient resource scheduling and optimization, improving system performance and security, and also reducing costs. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac6b17b985.png) + +- For details on the functions of the page buttons, please refer to the **[Tracing - Call Log](../tracing/call-log/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/08-pacp-strategy.md b/translate/translated/06-guide/02-ee-tenant/06-network/08-pacp-strategy.md new file mode 100644 index 00000000..7c862426 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/08-pacp-strategy.md @@ -0,0 +1,38 @@ +--- +title: PCAP Strategy +permalink: /guide/ee-tenant/network/pacp-strategy/ +--- + +> This document was translated by ChatGPT + +# PCAP Strategy + +The PCAP strategy supports setting policies or rules for capturing network packets. The strategy allows you to configure the network location for packet capture, the collector, filtering rules, payload truncation, etc., i.e., which types of packets should be captured and how to filter out unnecessary packets. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac6b204ac3.png) + +- **① Create New**: Supports creating a new PCAP strategy. For details, please refer to the **New Strategy** section. +- **② Enable/Disable**: Choose to enable or disable the current PCAP strategy. Once enabled, data filtering and capturing will be performed. +- **③ View Captured Traffic**: Click to jump to the PCAP download page to view the traffic data collected by this strategy. For details, please refer to the **[PCAP Download](./pcap-download/)** section. +- **④ Edit**: Modify the selected strategy. +- **⑤ Delete**: Delete the strategy. + +### Create New Strategy + +![Create New Strategy](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac6b31a676.png) + +- Name: Required. The name of the PCAP strategy. +- Network Location: Required. Select the network location where data will be captured. +- Collector: Select a supported collector based on the chosen network location. +- Capture Point Filter: Required. You can choose from computing resources, network resources, or container resources. + - For resources under different categories, you can further select specific resource information. For example, if you select an IP address under network resources, you need to enter the IP address to filter. +- VPC: Optional. Filter as needed. +- Protocol: Optional. Filter as needed. +- Port: Optional. Filter as needed. +- Peer: Disabled by default. Supports filtering peer data. + - Peer Filter: Required. Please refer to `Collector Filter` for input. + - Port: Optional. Filter as needed. +- Payload Truncation: Enter the size of the traffic to be truncated, in bytes. + - Default is 0, meaning no truncation. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/06-network/09-pcap-download.md b/translate/translated/06-guide/02-ee-tenant/06-network/09-pcap-download.md new file mode 100644 index 00000000..d13cdea6 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/06-network/09-pcap-download.md @@ -0,0 +1,22 @@ +--- +title: PCAP Download +permalink: /guide/ee-tenant/network/pcap-download/ +--- + +> This document was translated by ChatGPT + +# PCAP Download + +PCAP Download displays the data of enabled PCAP policies within a specified time range in both list and trend chart formats. + +## Overview + +![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230920650ac82daa46d.png) + +- **① PCAP Policy Dropdown**: The dropdown lists all PCAP policies, allowing you to select a specific policy. + - If no policy is selected, data for all PCAP policies will be displayed. +- **② Time**: Select the time range you want to view. If no PCAP policy was enabled during this period, no data will be shown. +- For page button functions, please refer to the **[Tracing - Call Log](../tracing/call-log/)** section. +- Log Details: Displays the captured traffic log information under the PCAP policy within the selected time range in a table format. + - Operations: + - Click Row: Click a data row to view its details in a right sliding panel. For usage details, please refer to the **[Right Sliding Panel - Flow Log Details](../tracing/right-sliding-box/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/07-metrics/01-overview.md b/translate/translated/06-guide/02-ee-tenant/07-metrics/01-overview.md new file mode 100644 index 00000000..f4a26572 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/07-metrics/01-overview.md @@ -0,0 +1,18 @@ +--- +title: Overview +permalink: /guide/ee-tenant/metrics/overview/ +--- + +> This document was translated by ChatGPT + +# Overview + +DeepFlow integrates with Prometheus data to display infrastructure status, CPU, memory, and other metrics in a list format. It currently supports displaying data for `Host` and `Container`. It also supports searching and viewing metric data information, as well as adding metric templates to specified data tables. + +- [Host](./host/) +- [Container](./container/) +- [Metrics Viewing](./metrics-viewing/) +- [Metric Summary](./metric-summary/) +- [Metrics Template](./metrics-template/) + +Note: When using `Host` or `Container`, you need to push Prometheus node_exporter data to DeepFlow. For the push method, refer to [Integrating Prometheus Data](../../../integration/input/metrics/prometheus/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/07-metrics/02-host.md b/translate/translated/06-guide/02-ee-tenant/07-metrics/02-host.md new file mode 100644 index 00000000..c3d9feeb --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/07-metrics/02-host.md @@ -0,0 +1,16 @@ +--- +title: Host +permalink: /guide/ee-tenant/metrics/host/ +--- + +> This document was translated by ChatGPT + +# Host + +Displays the host name, instance IP, operating system, CPU usage, MEM usage, and system load in a list format. + +![01-Host](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310caa83d06.png) + +Clicking a row will open a right-side sliding panel to view historical metrics. + +![02-Right-Side-Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310cab63d91.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/07-metrics/03-container.md b/translate/translated/06-guide/02-ee-tenant/07-metrics/03-container.md new file mode 100644 index 00000000..415d84ba --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/07-metrics/03-container.md @@ -0,0 +1,18 @@ +--- +title: Container +permalink: /guide/ee-tenant/metrics/container/ +--- + +> This document was translated by ChatGPT + +# Container + +Displays a list of container POD names, instance IPs, associated Nodes, namespaces, restart counts, statuses, and total runtime. + +![01-Container](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310caccb45f.png) + +Clicking a row will open a right-side panel to view historical metrics. + +![02-Right-Side-Panel](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023101965310cac1f0c1.png) + +**① Switch container dropdown**: Use this dropdown to quickly switch the container of the selected POD. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/07-metrics/04-metrics-viewing.md b/translate/translated/06-guide/02-ee-tenant/07-metrics/04-metrics-viewing.md similarity index 51% rename from translate/translated/06-guide/01-ee-tenant/07-metrics/04-metrics-viewing.md rename to translate/translated/06-guide/02-ee-tenant/07-metrics/04-metrics-viewing.md index 970da1ac..9dcd468d 100644 --- a/translate/translated/06-guide/01-ee-tenant/07-metrics/04-metrics-viewing.md +++ b/translate/translated/06-guide/02-ee-tenant/07-metrics/04-metrics-viewing.md @@ -7,12 +7,12 @@ permalink: /guide/ee-tenant/metrics/metrics-viewing/ # Metrics Viewing -On the metrics page, you can choose to view the status of metric data over a certain period of time. +On the metrics page, you can choose to view the status of metric data within a specified time range. ![Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240514664334ae9febc.png) - Usage: - - After filtering conditions in the search bar, you can add the metrics you want to query as needed - - The metrics viewing page supports adding multiple query filters -- For the use of search conditions, please refer to the chapter 【[Query](../query/overview/)】 -- For the use of line charts, please refer to the chapter 【[Charts - Line Chart](../dashboard/panel/line/)】 \ No newline at end of file + - After setting filter conditions in the search bar, you can add the metrics you want to query as needed + - The metrics viewing page supports adding multiple query filter conditions +- For the usage of search conditions, please refer to the **[Query](../query/overview/)** section +- For the usage of line charts, please refer to the **[Charts - Line Chart](../dashboard/panel/line/)** section \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/07-metrics/05-metric-summary.md b/translate/translated/06-guide/02-ee-tenant/07-metrics/05-metric-summary.md new file mode 100644 index 00000000..e995e198 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/07-metrics/05-metric-summary.md @@ -0,0 +1,15 @@ +--- +title: Metric Summary +permalink: /guide/ee-tenant/metrics/metric-summary/ +--- + +> This document was translated by ChatGPT + +# Metric Summary + +Switch the `Metric Set` to view the Metrics and Tag-related data stored in the corresponding data table. + +![2_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650bb8734ad16.png) + +- **① Metric Set**: Supports switching to search data tables +- **② Tab Switch**: Switch tabs to view the Metrics and Tags stored in the corresponding data table \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/07-metrics/06-metrics-template.md b/translate/translated/06-guide/02-ee-tenant/07-metrics/06-metrics-template.md new file mode 100644 index 00000000..8dbc9ce8 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/07-metrics/06-metrics-template.md @@ -0,0 +1,34 @@ +--- +title: Metrics Template +permalink: /guide/ee-tenant/metrics/metrics-template/ +--- + +> This document was translated by ChatGPT + +# Metrics Template + +A collection of metric sets created for different data tables, which can be used in the chart editing and query module. + +## Overview + +Displays the list of metric templates viewable by the current account, with support for editing, deleting, and other operations. Some data tables contain default templates, which are created by the system and cannot be edited or deleted. + +![00-Overview](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240514664334ae9febc.png) + +- **① Create Template**: Create a metrics template. For details, please refer to the **Create Metrics Template** section. +- **② Switch Data Table**: Switch the data table to view the corresponding metric templates under that table. +- **③ Delete**: Delete a metrics template. +- **④ Edit**: Edit a metrics template. +- **⑤ Expand Row**: Click a table row to quickly expand and view the metrics included in the template. For details, please refer to the figure below **01-Expand Row**. + +![01-Expand Row](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405146643341c20532.png) + +## Create Metrics Template + +![02-Create Metrics Template](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051466433419ccda4.png) + +- Template Name: Required. The name of the template being created. +- Team: Required. Select the team that can view the template. +- Data Table: Required. Select the data table to which the template will be added. +- Metrics: Select the metrics to be added. + - Supports setting aggregation operators, metric aliases, units, thresholds, etc. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/08-log/01-log.md b/translate/translated/06-guide/02-ee-tenant/08-log/01-log.md new file mode 100644 index 00000000..fb517b8c --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/08-log/01-log.md @@ -0,0 +1,26 @@ +--- +title: Logs +permalink: /guide/ee-tenant/log/log/ +--- + +> This document was translated by ChatGPT + +# Logs + +Logs display information related to the system and application processes. DeepFlow enhances log functionality through advanced log processing technologies and methods, improving the completeness and accuracy of data tracing and collection, and strengthening storage and display capabilities to meet enterprise application requirements. + +![01-Logs](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024061966729f8a3a45b.png) + +- **① Time Selector**: Supports time-based filtering queries. For usage details, please refer to [Dashboard Details - Time Selector](../dashboard/use/). +- **② Search Snapshot**: Supports saving search conditions as snapshots. For usage details, please refer to [Query - Search Snapshot](../query/history/). +- **③ Search Save and Settings**: + - Save Search: Supports quickly saving the current page's search conditions. For usage details, please refer to [Query - Search Snapshot](../query/history/). + - Settings: A collection of page setting operations. + - Database Fields: Supports viewing the tags and metrics from the data table used on the current page. + - Name Abbreviation/Full Display: Choose whether to display legend names in full or abbreviated form. +- **④ Service Search Box**: Supports searching or grouping by Tag. For usage details, please refer to [Query](../query/overview/). +- **⑤ Left Quick Filter**: Supports quick filtering by `Application Service` and `Log Level`. For usage details, please refer to [Query - Left Quick Filter](../query/left-quick-filter/). +- **⑥ Region Query**: Supports quickly switching the query data by region. +- **Table Operations**: + - Click Row: Clicking on a table row allows you to quickly open the right sliding panel to view related information of the corresponding application service. For usage details, please refer to [Tracing - Right Sliding Box](../tracing/right-sliding-box/). + - ⑦ Copy: Table content copy function. After clicking, you can choose the format of the content to copy. For configuration details, please refer to [Table](../dashboard/panel/table/). \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/09-alert/01-overview.md b/translate/translated/06-guide/02-ee-tenant/09-alert/01-overview.md new file mode 100644 index 00000000..ac3d7071 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/09-alert/01-overview.md @@ -0,0 +1,16 @@ +--- +title: Overview +permalink: /guide/ee-tenant/alert/overview/ +--- + +> This document was translated by ChatGPT + +# Overview + +Supports user-defined alert policies, allowing monitored data to be pushed to metric objects in real time, and also enabling users to view related information on the alert events page. + +DeepFlow’s alert feature consists of three main pages, which will be described in detail below. + +- [Alert Policy](./alert-policy/) +- [Push Endpoint](./push-endpoint/) +- [Alert Event](./alert-event/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/09-alert/02-alert-policy.md b/translate/translated/06-guide/02-ee-tenant/09-alert/02-alert-policy.md new file mode 100644 index 00000000..398248c2 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/09-alert/02-alert-policy.md @@ -0,0 +1,45 @@ +--- +title: Alert Policy +permalink: /guide/ee-tenant/alert/alert-policy/ +--- + +> This document was translated by ChatGPT + +# Alert Policy + +The formulation of an alert policy is used to identify and respond to abnormal conditions in programs or services, ensuring the normal operation of the system and the stability of business processes. +The Alert Policy page displays information about all alert policies in a list format. + +![00-总览.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566447b93e1025.png) + +- Number of Alerts: Displays the number of alert data generated after the corresponding alert policy takes effect. Supports clicking to jump to the [Alert Event](./alert-event/) page for further details. +- Status: Allows enabling or disabling the policy. +- Actions: + - Edit: Edit the corresponding alert policy, supporting modification of alert level and push endpoints. For usage details, please refer to the **Edit Alert Policy** section. + - Delete: Only supports deleting alert policies that are `disabled`. + +## Edit Alert Policy + +An alert policy requires configuration of three modules: Basic Information, Monitoring Configuration, and Notification Configuration, to generate the desired alert policy. + +![01-编辑告警策略.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566447b8697e3d.png) + +- Basic Information + - Alert Name: Required. Enter the corresponding alert name. + - Team: Required. Select the team or organization that can view the policy. + - Add Tags: Supports adding custom tags to the alert policy. + - Level: The importance level of the alert policy. + +![02-编辑告警策略.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051566447b880ecc2.png) + +- Monitoring Configuration + - Monitoring Frequency: The time interval between two data monitoring operations. + - Monitoring Interval: The time range for data queries each time the policy is executed. + - Options: `1 minute`, `5 minutes`, `15 minutes`, `30 minutes`, `1 hour` + - Monitoring Metric: Select the data metric to monitor. + - Event Level: Based on the set conditions, monitoring events can be classified into six levels: `Critical`, `Error`, `Warning`, `Recovery`, `Info`, and `No Data`. + - Recovery: When the results of consecutive X monitoring events do not meet any of the conditions for **Critical**, **Error**, **Warning**, or **No Data**, a `Recovery Event` is generated. + - Info: When enabled, if the monitoring event results do not meet any of the conditions for **Critical**, **Error**, **Warning**, **Recovery**, or **No Data**, an `Info Event` is generated. + - No Data: When enabled, if the monitoring event has no data, a `No Data Event` is generated. + - Notification Configuration + - Push Endpoint: Select the target(s) to push to. Multiple targets are supported. For configuration details, please refer to the **[Push Endpoint](./push-endpoint/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/09-alert/03-push-endpoint.md b/translate/translated/06-guide/02-ee-tenant/09-alert/03-push-endpoint.md new file mode 100644 index 00000000..8e2f56ab --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/09-alert/03-push-endpoint.md @@ -0,0 +1,100 @@ +--- +title: Push Endpoint +permalink: /guide/ee-tenant/alert/push-endpoint/ +--- + +> This document was translated by ChatGPT + +# Push Endpoint + +Push endpoints are systems or services used to receive and process alerts. Currently, four push methods are supported: Email push, HTTP push, Kafka push, PCAP policy, and Syslog push. + +The following sections introduce these five push methods in detail. + +## Email Push + +Sends alert events to a specified email address, allowing you to stay informed of alerts by checking your email. + +![Email Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230428644b76b451e05.png) + +- Create a new Email push: Fill in the relevant information. Once successfully created, it can be used when [creating an alert policy](./alert-policy/). +- List + - Associated alert policies: Click the number to jump to the **[Alert Policy](./alert-policy/)** page to view the alert policies using this push endpoint. + - Edit: Supports editing the push endpoint. + - Delete: Supports deleting the push endpoint. + +### Create a New Email Push + +![Create a New Email Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b62681d2f.png) + +- Email: Enter the email address to which alerts will be pushed. +- Push title: Optional, supports entering an email subject. +- For other fields, please refer to the **Create a New Kafka Push** section. + +## HTTP Push + +HTTP push sends data to a specified URL address via the HTTP protocol. + +![HTTP Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230428644b7a5c0c7bd.png) + +- For page button usage, please refer to the **Email Push** section. + +### Create a New HTTP Push + +![Create a New HTTP Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b6260915a.png) + +- Push method: Required. Supports `POST`, `PUT`, and `PATCH` methods, with `POST` as the default. +- Push URL: Required. Protocol names are case-insensitive. Supports HTTP and HTTPS protocols, with `HTTPS` supporting one-way authentication. + - Note: Supports Jinja template rendering, e.g., `http://10.0.0.1/{{policy_level}}` +- Header: Enter `HTTP` key-value pairs. +- For other fields, please refer to the **Create a New Kafka Push** section. + +## Kafka Push + +Kafka push supports sending alert events to Kafka. + +- For page button usage, please refer to the **Email Push** section. + +### Create a New Kafka Push + +![Create a New Kafka Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2024051666456bfb6ebdc.png) + +- Name: Required. Enter the name of the push endpoint. +- Team: Required. Select the team that can use this push endpoint. +- Broker address pool: Required. Input format `[address]:[port]`. Supports multiple entries separated by commas. +- Topic: Required. The Kafka topic for pushing alerts. Supports 1–256 printable characters. +- SASL: Optional. Authentication method. If `Plain` is selected, a username and password must be provided. +- Push content: Supports Jinja template rendering. For default push content, please refer to the parameter description. +- Configuration level: Select the alert event levels to receive. For alert event level descriptions, please refer to the **[Edit Alert Policy](./alert-policy/)** section. + - By default, all alert levels except `Info` will be pushed. +- Push cycle: Required. Within the push cycle, only one alert event will be pushed for the same monitored object under the same alert policy. +- Push frequency: The maximum number of times an alert event for the same monitored object under the same alert policy can be pushed. Exceeding this limit will stop further pushes. + +## PCAP Policy + +Supports adding alert policies to PCAP policies for alert monitoring via PCAP. + +![Create a New PCAP Policy](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20240516664573d598713.png) + +- For page button usage, please refer to the **Email Push** section. +- Create a new PCAP policy + - Associated PCAP policy: Required. Associates the PCAP policy with alert events. Alerts generated by monitoring can be downloaded in the associated PCAP policy. +- Enable PCAP policy: Select the alert event levels to push. If an alert of the selected level is generated, it will be pushed to the associated PCAP policy, and the PCAP policy will be automatically enabled. + - Note: By default, alerts of **Critical, Error, and Warning** levels will be pushed. +- Disable PCAP policy: Select the alert event levels to push. If an alert of the selected level is generated, the associated PCAP policy will be automatically disabled. + - Note: By default, **Recovery** alerts will be pushed. +- For other fields, please refer to the **Create a New Kafka Push** section. + +## Syslog Push + +Syslog push sends alert information to a log server via the Syslog protocol. It can notify operations personnel in real time of potential system failures or security incidents, helping them take timely action. + +- For page button usage, please refer to the **Email Push** section. + +### Create a New Syslog Push + +![Create a New Syslog Push](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645b6249798c.png) + +- Push destination: Required. Input format `[forwarding protocol]://[log server address]:[port]` + - Note: Supported protocols are `UDP` and `TCP`, with `UDP` as the default. Supported ports are `1–65535`, with `514` as the default. +- For other fields, please refer to the **Create a New Kafka Push** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/09-alert/04-alert-event.md b/translate/translated/06-guide/02-ee-tenant/09-alert/04-alert-event.md new file mode 100644 index 00000000..21b91d3c --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/09-alert/04-alert-event.md @@ -0,0 +1,17 @@ +--- +title: Alert Events +permalink: /guide/ee-tenant/alert/alert-event/ +--- + +> This document was translated by ChatGPT + +# Alert Events + +The Alert Events page displays events generated by alert policy monitoring. For example, when the system detects security vulnerabilities, abnormal behavior, or other situations that require user attention, it will generate corresponding alert events and provide detailed descriptions and recommendations. + +Users can filter and search based on multiple dimensions such as policy level, event level, monitored object, creator, monitored object, region, and event UID. + +![4_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650bf3357d79c.png) + +- You can use the `Column Options` feature in the table to add the event information you want to view. For details, please refer to the **[Tracing - Call Log](../tracing/call-log/)** section. +- Double-click a table row to display the detailed information of the corresponding alert event in a pop-up window. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/10-report/01-report.md b/translate/translated/06-guide/02-ee-tenant/10-report/01-report.md new file mode 100644 index 00000000..246dbaa0 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/10-report/01-report.md @@ -0,0 +1,47 @@ +--- +title: Reports +permalink: /guide/ee-tenant/report/report/ +--- + +> This document was translated by ChatGPT + +# Reports + +Reports can be created in a dashboard and scheduled to be pushed to users in an offline downloadable HTML format, recording the dashboard results within the report period. + +## Report Policies + +The Report Policies page displays all report policies in a list format. + +![01-report-forms.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540cc8a9690e.png) + +- **① Policy Name (Number of Reports)**: Click the policy name to navigate to the `Report Download` page to view all reports generated by this policy. +- **② Object**: Displays the name of the dashboard that generated the current report policy. Click to navigate to the corresponding dashboard. For details on dashboard usage, please refer to [Dashboard - Dashboard Details](../dashboard/use/). +- **③ Enable/Disable**: Start or stop report generation for the current policy. +- **④ Edit**: Allows modification of the current policy’s name and push email address. +- **⑤ Delete**: Delete the current policy. + +### Create a New Report Policy + +The following describes in detail how to create a report policy from a dashboard. + +- Step 1: Enter the dashboard for which you want to generate a report, click `Settings`, and select `Create Report Policy` to create a report. + +![02-dashboard.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540cc961795d.png) + +- Step 2: Configure the report policy according to your push requirements. + - Fill in the policy name, cycle, report format, statistical granularity, push email address, and other information. + - Note: Reports generated by the policy will be pushed to the email daily. + - For example, if the cycle is set to a weekly report, the report for the previous week (e.g., last Tuesday to this Tuesday) will be pushed daily. The report covers from 00:00 on the start date to 24:00 on the end date. Any modification to the report policy will take effect from the day it is made and will only affect subsequent reports, without impacting previously generated reports. The report email will include the report as an attachment, which can be downloaded by clicking. + +![03-add-report-forms.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540cca8c9511.png) + +## Report Download + +The Report Download page displays all generated reports in reverse chronological order (most recent first). + +![04-download-report-forms.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202310316540ccc1e1fec.png) + +- **① Dashboard Name**: Displays the name of the dashboard that generated the current report policy. Click to navigate to the corresponding dashboard. For details on dashboard usage, please refer to [Dashboard - Dashboard Details](../dashboard/use/). +- **② Download**: Allows downloading the current report as an HTML page. +- **③ Delete**: Delete the current report. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/01-overview.md b/translate/translated/06-guide/02-ee-tenant/11-resources/01-overview.md similarity index 68% rename from translate/translated/06-guide/01-ee-tenant/11-resources/01-overview.md rename to translate/translated/06-guide/02-ee-tenant/11-resources/01-overview.md index 35a89cab..12869d66 100644 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/01-overview.md +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/01-overview.md @@ -7,7 +7,7 @@ permalink: /guide/ee-tenant/resources/overview/ # Overview -In the resource module, DeepFlow supports viewing related resources and service information, such as resource pools, computing resources, network resources, network services, process resources, etc. Users can view according to their needs. +In the Resources module, DeepFlow supports viewing related resources and service information, such as resource pools, computing resources, network resources, network services, process resources, and more. Users can view them as needed. - [Summary](./summary/) - [Change Events](./resource-pool/) diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/02-summary.md b/translate/translated/06-guide/02-ee-tenant/11-resources/02-summary.md new file mode 100644 index 00000000..dd5f368d --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/02-summary.md @@ -0,0 +1,15 @@ +--- +title: Summary +permalink: /guide/ee-tenant/resources/summary/ +--- + +> This document was translated by ChatGPT + +# Summary + +The Summary page displays the distribution of network and computing resources. Information for each resource is presented in pie charts. + +![Summary](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023042464463e64633af.png) + +- **Network Resources**: Provides statistical visualization of the distribution of VPCs, subnets, and public IPs across cloud platforms. +- **Computing Resources**: Displays the distribution of cloud platforms where cloud servers are located, their running status, and the distribution status of corresponding collectors. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/03-resource-changes.md b/translate/translated/06-guide/02-ee-tenant/11-resources/03-resource-changes.md new file mode 100644 index 00000000..afecf76e --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/03-resource-changes.md @@ -0,0 +1,16 @@ +--- +title: Change Events +permalink: /guide/ee-tenant/resources/resource-changes/ +--- + +> This document was translated by ChatGPT + +# Change Events + +On the Change Events page, users can view events occurring in the system in real time, such as usage, file creation, file deletion, and more. Users can filter event types as needed and quickly locate events of interest. + +Users can view events that occurred within a specific time range, or filter by user or resource to better understand the details and context of the events. + +![1_1.png](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230921650bbbf54f94c.png) + +- For details on the functions of the page buttons, please refer to the **[Tracing - Call Log](../tracing/call-log/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/04-resource-pool.md b/translate/translated/06-guide/02-ee-tenant/11-resources/04-resource-pool.md new file mode 100644 index 00000000..e258f8b3 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/04-resource-pool.md @@ -0,0 +1,54 @@ +--- +title: Resource Pool +permalink: /guide/ee-tenant/resources/resource-pool/ +--- + +> This document was translated by ChatGPT + +# Resource Pool + +In a cloud computing environment, the resource pool organizes and displays resource collection information at three levels: cloud platform, region, and availability zone. By organizing cloud environment resource pools, a cloud computing platform can better manage its internal resources, improve resource utilization and efficiency, and meet users’ personalized needs. + +The following sections introduce each of the three organizational levels. + +## Cloud Platform + +The cloud platform displays the resource pools and related service information data within the cloud computing environment’s cloud platform, and supports functions such as creating a new cloud platform, configuring synchronization intervals, and making modifications. The page displays the data information of each added cloud platform in a table format, such as type, number of regions, number of availability zones, affiliated container clusters, resource synchronization controllers, status, and more. + +![01-Cloud Platform](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230424644643e1209c0.png) + +- **① Action Buttons**: + - Create Cloud Platform: Supports creating a new entry for an already connected cloud platform + - Export CSV: Supports downloading the data + - Configure Sync Interval: Sets the synchronization rate for the cloud platform + - Sync Time Statistics: Displays synchronization time data for all added cloud platforms on the page in a line chart +- **② Expand Row**: Click the row header to display more detailed data information for the corresponding cloud platform +- **③ Number of Regions**: Click to navigate to the page showing the corresponding region information, see the **Region** section +- **④ Number of Availability Zones**: Click to navigate to the page showing the corresponding availability zone information, see the **Availability Zone** section +- **⑤ Enable/Disable**: Start or stop synchronizing data for the cloud platform +- **⑥ Edit**: Edit the data in the row +- **⑦ Delete**: Delete the data in the row + +## Region + +The region displays the geographical location of data centers within the cloud platform. You can view related information about the region, such as the number of availability zones, VPCs, subnets, cloud servers, and PODs. It also supports creating new entries, modifying, and exporting data. + +![02-Region](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230424644650cec4b7f.png) + +- **① Action Buttons**: + - Create Region: Supports creating a new region + - Export CSV: Supports downloading the data +- **② Number of Availability Zones**: Click to navigate to the page showing the availability zones contained in the corresponding region, see the **Availability Zone** section +- **③ Number of VPCs**: Click to navigate to the page showing the VPCs contained in the corresponding region, see **[Network Resources - VPC](./network-resources/)** +- **④ Number of Subnets**: Click to navigate to the page showing the subnets contained in the corresponding region, see **[Network Resources - Subnet](./network-resources/)** +- **⑤ Number of Cloud Servers**: Click to navigate to the page showing the cloud servers contained in the region, see **[Compute Resources - Cloud Server](./network-resources/)** +- **⑥ Number of PODs**: Click to navigate to the page showing the container PODs contained in the region, see **[Container Resources - Container POD](./network-resources/)** +- **⑦ Edit**: Edit the data in the row + +## Availability Zone + +The availability zone displays data center information within the cloud platform, including the region and cloud platform it belongs to, the number of cloud servers, and the number of PODs. It supports creating new entries, modifying, and exporting data. + +![03-Availability Zone](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230425644783b74e992.png) + +- For page button usage, see the **Region** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/05-computing-resources.md b/translate/translated/06-guide/02-ee-tenant/11-resources/05-computing-resources.md new file mode 100644 index 00000000..9259c4ef --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/05-computing-resources.md @@ -0,0 +1,30 @@ +--- +title: Computing Resources +permalink: /guide/ee-tenant/resources/computing-resources/ +--- + +> This document was translated by ChatGPT + +# Computing Resources + +The Computing Resources page allows you to view the physical or virtual devices used for processing and executing computing tasks. You can view information for cloud servers and host machines separately. + +## Cloud Server + +Cloud servers allow users to deploy and run various applications. The Cloud Server page displays related information. + +![Cloud Server](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304256447a5dd95922.png) + +- **① Action Buttons**: + - Create Cloud Server: Create a new cloud server + - Export CSV: Download the data + - All NICs: Display all network interface card information in a list, including IP, MAC, name, subnet, associated cloud server, etc., and supports CSV export +- **② Refresh Cache**: Refresh the cache for faster data access +- **③ VPC**: Click to navigate to the page showing the VPC information of the corresponding cloud server. See the [Network Resources - VPC](./network-resources/) section +- **④ Collector Status**: View collector-related information for the cloud server, such as basic information, configuration, monitoring data, and operation logs +- **⑤ Edit**: Only supports name modification +- **⑥ NIC List**: View the NIC information under the current cloud server + +## Host Machine + +A host machine is a physical server running virtualization software on an operating system. On the Host Machine page, you can view related information such as region, availability zone, related IP information, CPU, memory, NICs, collectors, and more. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/06-network-resources.md b/translate/translated/06-guide/02-ee-tenant/11-resources/06-network-resources.md new file mode 100644 index 00000000..350b4af3 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/06-network-resources.md @@ -0,0 +1,49 @@ +--- +title: Network Resources +permalink: /guide/ee-tenant/resources/network-resources/ +--- + +> This document was translated by ChatGPT + +# Network Resources + +The Network Resources page allows you to view information about resources in the current network architecture, including VPCs, subnets, routers, DHCP gateways, and IP addresses. Each network resource has different functions and purposes, and can be used to build a complete network architecture. + +## VPC + +A VPC is a virtual network environment. On the VPC page, you can view relevant data information within the virtual environment. + +![01-VPC](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448916ab2058.png) + +- For the usage of page buttons, refer to the section **[Resource Pool - Region](./network-resources/)** + +## Subnet + +A subnet is a logical partition within a VPC, corresponding to an actual network address range. Instances within a subnet can connect to other subnets or the public network through the VPC's router. + +![02-Subnet](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448943337299.png) + +- For the usage of page buttons, refer to the section **[Resource Pool - Region](./network-resources/)** + +## Router + +A router is responsible for transmitting requests from one network to another. On the Router page, you can view information about all routers. + +![03-Router](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2023042664489e7cde5f0.png) + +- All Routing Rules: Allows you to view all routing tables, which are tables containing address paths pointing to specific destinations, including destination IP, next hop type, next hop, and associated router. +- Actions + - View Routing Rules: View the routing table of the current router. +- For other page button usage, refer to the section **[Resource Pool - Region](./network-resources/)** + +## DHCP Gateway + +A DHCP gateway supports automatic IP address allocation. On the DHCP Gateway page, you can view information about all DHCP gateways, such as region, VPC, IP, cloud platform, deletion time, etc. + +## IP Address + +An IP address is a numerical address used to identify each device in a network. + +![04-IP Address](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448be4281264.png) + +- For other page button usage, refer to the section **[Resource Pool - Region](./network-resources/)** \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/07-network-services.md b/translate/translated/06-guide/02-ee-tenant/11-resources/07-network-services.md new file mode 100644 index 00000000..ee610597 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/07-network-services.md @@ -0,0 +1,46 @@ +--- +title: Network Services +permalink: /guide/ee-tenant/resources/network-services/ +--- + +> This document was translated by ChatGPT + +# Network Services + +Network services display relevant data information for different components in a cloud computing network architecture, including security groups, NAT gateways, load balancers, peering connections, and cloud enterprise networks. + +## Security Group + +A security group is a type of virtual network security mechanism in cloud computing that can control inbound and outbound traffic of cloud instances, thereby protecting the security of cloud instances and the cloud network. + +![01-Security Group](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448d05507924.png) + +- All security group rules: Supports viewing all security group rules, including direction, IP type, protocol type, local port range, remote port range, and other information, and also supports exporting to CSV. +- Operations + - View security group rules: Supports viewing all rules of the selected security group. +- For the usage of other buttons on the page, please refer to the **[Resource Pool - Region](./network-resources/)** section. + +## NAT Gateway + +A NAT gateway can perform network address translation between different subnets and enable communication between private network instances and the Internet. On the NAT gateway page, you can view and download related information such as public IP, region, VPC, number of gateway rules, cloud platform, etc., and also view and download gateway rule information. + +- For the usage of page buttons, please refer to the **[Resource Pool - Region](./network-resources/)** section. +- + +## Load Balancer + +In cloud computing, a load balancer distributes traffic from virtual machines to multiple virtual machines by adjusting the traffic, thereby achieving load balancing and improving the availability and scalability of applications. On the load balancer page, you can view and download related information such as IP, region, VPC, number of load balancers, cloud platform, etc., and also view and download rule information. + +- For the usage of page buttons, please refer to the **[Resource Pool - Region](./network-resources/)** section. + +## Peering Connection + +A peering connection is a local network connection service that supports establishing virtual private network interconnections between geographically different VPCs. On the peering connection page, you can view related information such as local region, local VPC, peer region, peer VPC, cloud platform, etc., and also create new peering connections and export to CSV. + +- For the usage of page buttons, please refer to the **[Resource Pool - Region](./network-resources/)** section. + +## Cloud Enterprise Network + +A cloud enterprise network is a cloud computing solution that provides interconnection between private networks and public networks, and can also achieve network interconnection between multiple different cloud providers, improving the security, stability, and reliability of data communication. On the cloud enterprise network page, you can view related information such as associated instances, cloud platform, etc., and also export to CSV. + +- For the usage of page buttons, please refer to the **[Resource Pool - Region](./network-resources/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/08-storage-services.md b/translate/translated/06-guide/02-ee-tenant/11-resources/08-storage-services.md new file mode 100644 index 00000000..a0894c54 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/08-storage-services.md @@ -0,0 +1,18 @@ +--- +title: Storage Services +permalink: /guide/ee-tenant/resources/storage-services/ +--- + +> This document was translated by ChatGPT + +# Storage Services + +Storage services are a fundamental offering provided by cloud computing platforms, helping users store and manage data and other objects, while delivering high availability, reliability, and elastic scalability. Within storage services, you can view information for both Cloud Database RDS and Cloud Database Redis. + +## Cloud Database RDS + +Cloud Database RDS, as a highly available, high-performance, and secure managed database service, provides users with convenient and flexible storage services along with backup and recovery capabilities. It meets enterprise needs for data storage and management, while taking on related administrative tasks to improve operational efficiency. On the Cloud Database RDS page, you can view and download relevant information such as region, availability zone, VPC, subnet, private IP, public IP, status, database platform, cloud platform, and more. + +## Cloud Database Redis + +Cloud Database Redis, as a high-speed caching database service, supports multiple data structures, significantly improving data read/write speed and performance, enhancing system performance and response time. It can be used to meet data caching needs and other functions in various scenarios. On the Cloud Database Redis page, you can view and download relevant information, with the list content consistent with that in the **Cloud Database RDS** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/09-container-resources.md b/translate/translated/06-guide/02-ee-tenant/11-resources/09-container-resources.md new file mode 100644 index 00000000..973191d9 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/09-container-resources.md @@ -0,0 +1,56 @@ +--- +title: Container Resources +permalink: /guide/ee-tenant/resources/container-resources/ +--- + +> This document was translated by ChatGPT + +# Container Resources + +Container resources include multiple dimensions of resources. Through proper configuration and limitation, resources can be allocated and managed reasonably to ensure that applications can run stably and reliably in a container environment. In container resources, you can view information on container clusters, namespaces, container nodes, Ingress, container services, workloads, ReplicaSets, and container PODs respectively. + +## Container Cluster + +A container cluster is a group of connected containers that achieve high availability and scalability by sharing network and storage resources. On the container resources page, you can view and download related information such as management platform, availability zone, cloud platform, VPC, number of nodes, etc. + +## Namespace + +A namespace is used to provide an isolated runtime environment for containers to avoid conflicts between them. On the namespace page, you can view and download related information such as availability zone, number of PODs, number of ReplicaSets, clusters, etc. + +## Container Node + +A container node refers to a physical or virtual server that runs container instances in a container orchestration system. On the container node page, you can view and download related information such as availability zone, cloud server, type, status, IP, internal routing IP, CPU, memory, cluster, number of PODs, collectors, etc. + +## Ingress + +Ingress is a way to expose HTTP and HTTPS services in a K8s cluster. This is achieved by creating an API object in the cluster, which routes public network traffic to the corresponding service. + +![01-Ingress](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448dfd092f9a.png) + +- All forwarding rules: Supports viewing and downloading all forwarding rules. The list includes protocol type, domain name, path, service, service PORT, and associated Ingress. +- Actions: + - View forwarding rules: You can view and download the forwarding rules under the current Ingress. +- For the usage of other buttons on the page, please refer to the **[Resource Pool - Region](./network-resources/)** section. + +## Container Service + +A container service is a cloud-based service that packages applications and services into one or more containers and deploys them to the cloud, enabling them to run anywhere. This simplifies the deployment and management process of applications. + +![02-Container Service](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448e6b382c9a.png) + +- All port mappings: Supports viewing and downloading all mappings of application ports inside containers to the host (node) ports, displaying protocol type, node PORT, service PORT, container PORT, and associated service, and also supports downloading. +- Actions: + - View port mappings: You can view and download the port mapping list of the current container. +- For the usage of other buttons on the page, please refer to the **[Resource Pool - Region](./network-resources/)** section. + +## Workload + +A workload refers to applications and services hosted on a cloud platform, including multiple components, services, processes, and containers. They can run based on container or virtualization technology to ensure availability and scalability. On the workload page, you can view and download related information such as availability zone, type, number of running PODs, number of desired PODs, number of ReplicaSets, namespace, cluster, K8s label, etc. + +## ReplicaSet + +A ReplicaSet is a controller in a K8s cluster for high availability, elasticity, and automated replica management. It ensures the availability of Pods and automatically adjusts the number of PODs when needed. On the ReplicaSet page, you can view and download related information such as availability zone, number of running PODs, number of desired PODs, workload, namespace, cluster, K8s label, etc. + +## Container POD + +A container is a way to package applications and dependencies, while a POD is a collection of multiple containers that can share resources and work together. On the container POD page, you can view and download related information such as service, MAC, IP, status, namespace, cluster, etc. \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/11-resources/10-process-resources.md b/translate/translated/06-guide/02-ee-tenant/11-resources/10-process-resources.md similarity index 58% rename from translate/translated/06-guide/01-ee-tenant/11-resources/10-process-resources.md rename to translate/translated/06-guide/02-ee-tenant/11-resources/10-process-resources.md index ca64a089..70c34aba 100644 --- a/translate/translated/06-guide/01-ee-tenant/11-resources/10-process-resources.md +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/10-process-resources.md @@ -7,8 +7,8 @@ permalink: /guide/ee-tenant/resources/process-resources/ # Process -A process is an instance of a program that is currently running on a computer. The process page supports viewing and downloading related data information. +A process is an instance of a program that is currently running on a computer. On the process page, you can view and download related data and information. ![Process](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266448fdf60b29f.png) -- For page button usage, please refer to the section 【[Resource Pool - Region](./network-resources/)】 \ No newline at end of file +- For instructions on using the page buttons, please refer to the **[Resource Pool - Region](./network-resources/)** section. \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/11-resources/11-other-resources.md b/translate/translated/06-guide/02-ee-tenant/11-resources/11-other-resources.md new file mode 100644 index 00000000..95cf46e9 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/11-resources/11-other-resources.md @@ -0,0 +1,33 @@ +--- +title: Other Resources +permalink: /guide/ee-tenant/resources/other-resources/ +--- + +> This document was translated by ChatGPT + +# Other Resources + +Other resources include information related to other resources in the network or system, such as physical network elements, network locations, physical links, and legends. + +## Physical Network Elements + +Physical network elements refer to hardware devices in the network, including routers, switches, firewalls, gateways, etc. These devices enable data communication and transmission by connecting and exchanging network traffic. On the Physical Network Elements page, you can view and download related information such as region and type. It also supports creating, modifying, and deleting physical network elements. + +## Network Locations + +Network locations refer to the sources or destinations of information collected during the information gathering and analysis process. On the Network Locations page, you can view and download related data such as region, data tags, type, VLAN tags, source IP, interface name, sampling rate, etc. It also supports creating, modifying, and deleting network locations. + +## Physical Links + +Physical links refer to the physical lines connecting network devices, which have a significant impact on the stability, speed, and reliability of network transmission. + +![01-Physical Link](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266449023482d95.png) + +- View Topology: Displays the logical dependency relationships of all physical links in the current list in the form of a topology diagram. +- For page button usage, please refer to the section **[Resource Pool - Region](./network-resources/)**. + +## Legends + +Legends represent related resources in the form of icons. All legends are displayed in a list, and creating and modifying legends is supported. + +![02-Legend](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202304266449034569faf.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/12-system/01-overview.md b/translate/translated/06-guide/02-ee-tenant/12-system/01-overview.md new file mode 100644 index 00000000..adebffff --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/12-system/01-overview.md @@ -0,0 +1,15 @@ +--- +title: Overview +permalink: /guide/ee-tenant/system/overview/ +--- + +> This document was translated by ChatGPT + +# Overview + +The system module allows users to view information such as agents, data nodes, accounts, and operation logs. + +- [Agent](./agent/) +- [Data Node](./data-node/) +- [Account Management](./account-management/) +- [Operation Log](./operation-log/) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/12-system/02-agent.md b/translate/translated/06-guide/02-ee-tenant/12-system/02-agent.md new file mode 100644 index 00000000..a2a9be3b --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/12-system/02-agent.md @@ -0,0 +1,89 @@ +--- +title: Collector +permalink: /guide/ee-tenant/system/agent/ +--- + +> This document was translated by ChatGPT + +# Collector + +The DeepFlow Collector is a tool used to collect network and application performance data. It supports parsing various TraceID and SpanID formats in protocols such as HTTP and Dubbo. + +The following introduces the collector module. + +## List + +The collector list displays the installation, deployment, and running status of collectors, and also supports batch operations on collectors. + +![01-List](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4a708edd.png) + +- Top row action buttons: + - Enable: Select multiple collectors to perform batch enable operations + - Disable: Select multiple collectors to perform batch disable operations + - Register: Select multiple collectors to perform batch registration operations + - Add to Collector Group: Select multiple collectors to perform batch add-to-group operations + - Export CSV: Select multiple collectors to perform batch export to CSV operations +- Collector list + - Name: Click to go to the collector details page to view collector information + - Basic Information: Displays the current basic information of the collector, its environment configuration, status information, and the environment in which it is running + - Configuration Information: Displays detailed information about the configuration of the collector group to which the collector belongs + - Monitoring Data: Displays all monitoring charts on the current collector details page + - Runtime Logs: In multi-region deployment scenarios, supports viewing runtime logs of collectors in non-primary regions. Displays all logs of the collector recorded in ES, with WARN logs highlighted in yellow and ERR logs highlighted in red + - Group: The group to which the collector belongs. Click to go to the **Group** page. Collectors without a specified group belong to the default group. Please refer to the **Group** section for details + - Type: Displays the current running environment of the collector + - KVM: The Trident process of the collector runs on the host machine (e.g., KVM) + - Container-V / Container-P: The Trident process of the collector runs as a DaemonSet on each container node (K8S Node) + - ESXi: The Trident process of the collector runs on a dedicated VM on vSphere ESXi, collecting mirrored traffic from all business VMs on ESXi + - Dedicated Server: The Trident process of the collector runs on a dedicated server, collecting mirrored traffic from physical switches + - Workload-V: The Trident process of the collector runs inside a business VM + - Workload-P: The Trident process of the collector runs inside a business bare-metal server + - Tunnel Decapsulation: The Rosen process of the collector runs on an independent server to remove tunnel encapsulation from distributed traffic + - Architecture: System information of the collector's running environment + - Operating System: System information of the collector's running environment + - Control IP: The IP address used by the collector to communicate with the controller + - Control MAC: The MAC address used by the collector to communicate with the controller + - Status: Displays the current status of the collector, including Unregistered / Running / Disconnected / Disabled + - Exception: Displays a red exclamation mark when there are exceptions during collector operation. Currently supported exceptions include: + - Self-check failed: Less than 100MB of free disk space for logs + - Self-check failed: Insufficient available memory + - Distribution circuit breaker triggered + - Distribution traffic reached rate limit + - Gateway ARP to distribution point not found + - Gateway ARP to data node not found + - Software Version: Displays the version number of the Trident/Agent software, used for troubleshooting and upgrade guidance + - Start Time: The time when the collector process was started + - Controller: The IP address of the controller from which the collector requests policies (also the target data node IP address for sending monitoring information). Click to go to the **Controller List** page. For usage details, please refer to the **Controller** section + - Controller Sync Time: The last time the collector synchronized cloud platform information with the controller + - Data Node: The target data node to which the collector sends data. Click to go to the **Data Node** page. For usage details, please refer to the **Data Node** section + - Current Access Data Node: When the data node is behind an SLB and serving the collector, this field displays the actual data node IP the collector is requesting + - Data Node Communication Time: The last communication time between the collector and the data node. Note that updates to this value may be delayed due to caching + - Actions: Enable/Disable, Delete + - Enable/Disable: Enable or disable the collector + - Delete: Delete unused collectors + +## Group + +Displays information about collector groups in list form, such as the number of collectors included, the number of disabled collectors, and the number of unregistered collectors. + +![02-Group](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4c187e7f.png) + +- Displays in list form the number of collectors in the group, the number of disabled collectors, and the number of unregistered collectors. Click the numbers to go to the collector page for details + - Supports creating new collector groups, registering, disabling, enabling, and deleting +- Collector Group: Groups hosts/cloud servers of the same type together for unified management + - Default: If the user does not assign a custom group to a collector, it is placed in the default group + - The default group cannot be modified or deleted. The collector belongs to the last group it was added to +- Note: For collector groups created by the platform, users do not have permission to register collectors, edit, disable, etc. + +## Configuration + +Displays detailed information about collector groups in list form, such as CPU limits, memory limits, packet capture rate limits, distribution rate limits, distribution circuit breaker thresholds, capture interfaces, and more. + +![03-Configuration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4d1b64aa.png) + +- Click a row to further display detailed information about the current collector group, such as resource limits, basic configuration parameters, universal map configuration parameters, packet distribution configuration parameters, basic functions, universal map feature switches, and packet distribution feature switches + +## Statistics + +Displays current collector-related status data in chart form, such as total capture traffic, total distribution traffic, collector CPU usage, collector memory usage, running environment load, packet loss count, cloud server capture traffic, and cloud server distribution traffic. + +![04-Statistics](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673d4e252f7f.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/12-system/03-data-node.md b/translate/translated/06-guide/02-ee-tenant/12-system/03-data-node.md new file mode 100644 index 00000000..e52f22f3 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/12-system/03-data-node.md @@ -0,0 +1,30 @@ +--- +title: Data Node +permalink: /guide/ee-tenant/system/data-node/ +--- + +> This document was translated by ChatGPT + +# Data Node + +Displays the configuration information of database-related data tables used on the page in a list format. + +![Data Node](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673ddda82472.png) + +- **Create Data Table**: Users can create new custom data tables based on existing data sources, supporting up to 10 data tables. The creation information is roughly as follows: + - Name: Supports Chinese, English, numbers, and underscores, with a maximum length of 10 characters + - Data Table Set: Supports user-created data table sets, in the format of database.data_table_set + - Options: flow_metrics.vtap_flow*, flow_metrics.vtap_app* + - Original Time Granularity: The original time granularity of the `data table set`. The system currently supports 1m (one minute) and 1s (one second) by default + - Time Granularity: The data table will be aggregated and calculated based on the selected time granularity + - Options: 1h (one hour), 1d (one day) + - Retention Period: Set the data retention time + - Additive Metric Aggregation: If the original data table uses sum, only sum can be selected; if the original data table uses max/min, only max/min can be selected + - Non-additive Metric Aggregation: todo +- **Edit**: Supports modifying the retention period of the data table +- **Delete**: Only supports deleting user-defined new data sources +- **Note**: + - The system has 15 default basic data tables, which cannot be deleted + - For metric-type (flow*metrics*_) and log-type (flow*log*_) data tables, minute-granularity data is retained for 1 week by default, and second-granularity data is retained for 1 day by default. Expired data is automatically deleted, and users can set an appropriate retention period accordingly + - PCAP data is retained for 1 week by default. Data older than 1 week will be automatically deleted, and users can set an appropriate retention period accordingly + - Monitoring data in the system is retained for 1 week by default. Data older than 1 week will be automatically deleted, and users can set an appropriate retention period accordingly \ No newline at end of file diff --git a/translate/translated/06-guide/01-ee-tenant/12-system/04-account-management.md b/translate/translated/06-guide/02-ee-tenant/12-system/04-account-management.md similarity index 71% rename from translate/translated/06-guide/01-ee-tenant/12-system/04-account-management.md rename to translate/translated/06-guide/02-ee-tenant/12-system/04-account-management.md index 68746316..175ceefd 100644 --- a/translate/translated/06-guide/01-ee-tenant/12-system/04-account-management.md +++ b/translate/translated/06-guide/02-ee-tenant/12-system/04-account-management.md @@ -7,6 +7,6 @@ permalink: /guide/ee-tenant/system/account-management/ # Account Management -You can view the basic information of the current account and also support changing the account password. +You can view the basic information of the current account, and also change the account password. ![Account Management](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202406206673deb1234e2.png) \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/12-system/05-operation-log.md b/translate/translated/06-guide/02-ee-tenant/12-system/05-operation-log.md new file mode 100644 index 00000000..30c34a56 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/12-system/05-operation-log.md @@ -0,0 +1,13 @@ +--- +title: Operation Log +permalink: /guide/ee-tenant/system/operation-log/ +--- + +> This document was translated by ChatGPT + +# Operation Log + +Displays the logs related to the current system and user operations in a list format. + +- Log Levels: `ERROR`, `WARN`, `INFO` +- Log Types: `User Login`, `User Operation`, `System Module`, `System Event`, `System Maintenance` \ No newline at end of file diff --git a/translate/translated/06-guide/02-ee-tenant/13-configuration/01-settings.md b/translate/translated/06-guide/02-ee-tenant/13-configuration/01-settings.md new file mode 100644 index 00000000..7ebf9ac8 --- /dev/null +++ b/translate/translated/06-guide/02-ee-tenant/13-configuration/01-settings.md @@ -0,0 +1,43 @@ +--- +title: Settings +permalink: /guide/ee-tenant/configuration/settings/ +--- + +> This document was translated by ChatGPT + +# Settings + +The Settings module supports editing preferences, viewing platform information, and more. + +## Preferences + +Supports configuring usage preferences for the search box on pages. + +### Search Box Configuration + +The search box configuration only applies to pages under **Events**, **Applications**, and **Network**. + +![Search Box Configuration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645a4bfcef96.png) + +- Follow system settings: When selected, the configuration is consistent with the default system settings. To adjust settings manually, simply uncheck this option. +- **Page initial load**: Whether to query data when the page is loaded for the first time. + - Do not trigger search: No data is displayed when the page first loads. You need to click the **Search** button or add query conditions to display data. + - Search with default conditions: `Default system setting`. When the page loads, data is queried and displayed simultaneously. +- **Search trigger method**: Set how search queries are triggered. + - Trigger instantly: `Default system setting`. When search conditions change, the query is triggered immediately. + - Trigger by clicking the **Search** button: When search conditions change, you need to click the **Search** button to trigger the query. +- **Default search box type**: For `Path`-type pages, set the default display form of the search box. + - Simplified search: `Default system setting`. For details, see **[Resource Search Box](../query/service-search/)**. + - One-way path: For details, see **[Path Search Box](../query/path-search/)**. + - Two-way path: For details, see **[Path Search Box](../query/path-search/)**. +- **Default search box content**: The search box can be set to quick search mode. + + - Free search: For details, see **[Resource Search Box](../query/service-search/)**. + - Container search: For details, see **[Resource Search Box](../query/service-search/)**. + - Process search: `Default system setting`. For details, see **[Resource Search Box](../query/service-search/)**. + +## Platform Information + +Platform information allows you to view the current system version number, feedback email, vendor information, and more. + +![Platform Information](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202405166645829f09cbb.png) \ No newline at end of file diff --git a/translate/translated/07-configuration/README.md b/translate/translated/07-configuration/README.md index 7b0c5993..b5c8502f 100644 --- a/translate/translated/07-configuration/README.md +++ b/translate/translated/07-configuration/README.md @@ -1,4 +1,8 @@ ---- -permalink: /configuration ---- -# configuration +--- +permalink: /zh/configuration +--- + +# Configuration Manual +> This document was translated by ChatGPT + +> This document was translated by ChatGPT diff --git a/translate/translated/08-integration/01-process/01-wasm-plugin.md b/translate/translated/08-integration/01-process/01-wasm-plugin.md index eb23df40..de6fbb99 100644 --- a/translate/translated/08-integration/01-process/01-wasm-plugin.md +++ b/translate/translated/08-integration/01-process/01-wasm-plugin.md @@ -7,42 +7,47 @@ permalink: /integration/process/wasm-plugin # About the Wasm Plugin System -The Wasm plugin system implements some user-defined functions by calling the Wasi Export Function at fixed points. We provide some examples in [this repository](https://github.com/deepflowio/deepflow-wasm-go-sdk/tree/main/example), through which you can see what functionalities the current DeepFlow Wasm Plugin can achieve: +The Wasm plugin system implements user-defined functions by invoking Wasi Export Functions at fixed points. We provide some examples in [this repository](https://github.com/deepflowio/deepflow-wasm-go-sdk/tree/main/example), which can help you understand the current capabilities of the DeepFlow Wasm Plugin: -| Category | Directory | Description | -| --------------- | ------------------- | --------------------------------- | -| Enhance Known Protocols | http | Parse JSON over HTTP | -| | http_status_rewrite | Parse JSON over HTTP | +| Category | Directory | Description | +| -------------- | ------------------- | ----------------------------------- | +| Enhance known protocols | http | Parse JSON over HTTP/HTTP2/gRPC | +| | http_status_rewrite | Parse JSON over HTTP/HTTP2/gRPC | | | dubbo | Parse JSON over Dubbo | | | nats | Parse Protobuf (nRPC) over NATS | | | zmtp | Parse Protobuf over ZMTP | -| Parse as New Protocol | krpc | Parse Protobuf over TCP | +| Treat as new protocol | krpc | Parse Protobuf over TCP | | | go_http2_uprobe | Parse Protobuf over HTTP2 | | | dns | Demonstrate how to parse DNS as a new protocol | -For more on Wasm Plugin development, you can also refer to this blog post: [使用 DeepFlow Wasm 插件实现业务可观测性](https://www.deepflow.io/blog/zh/035-deepflow-enabling-zero-code-observability-for-applications-by-webAssembly/). +For developing Wasm Plugins, you can also refer to this blog post: [使用 DeepFlow Wasm 插件实现业务可观测性](https://www.deepflow.io/blog/zh/035-deepflow-enabling-zero-code-observability-for-applications-by-webAssembly/). + +For HTTP2 and gRPC protocols, deepflow-agent already has built-in full header field parsing capabilities, and you can configure deepflow-agent to collect specific header fields via agent-group-config. Therefore, the HTTP2/gRPC Wasm Plugin only needs to parse the Payload. Note that there are two collection methods: cBPF/eBPF-kprobe (compressed Header + raw Payload) and eBPF-uprobe (raw Header + raw Payload), and the plugin writing method differs: + +- For data collected via cBPF/eBPF-kprobe, write the plugin by referring to the `http` plugin in the table above to parse the Payload. +- For data collected via eBPF-uprobe, currently only supported as a new protocol to be re-parsed in the Plugin, refer to the `go_http2_uprobe` plugin in the table above (enhancement support is still under development). # Golang SDK Instructions -Currently, only the Golang SDK is provided, with more languages to be supported in the future. The Golang SDK requires tinygo for compilation. Below is a brief explanation of how to quickly develop a plugin using Golang. +Currently, only the Golang SDK is provided, with more languages to be supported in the future. The Golang SDK requires tinygo for compilation. Below is a brief guide on how to quickly develop a plugin in Golang. ```go package main import ( "github.com/deepflowio/deepflow-wasm-go-sdk/sdk" - _ "github.com/wasilibs/nottinygc" // Use nottinygc as an alternative memory allocator for TinyGo compiling WASI, as the default allocator has performance issues in large data scenarios + _ "github.com/wasilibs/nottinygc" // Use nottinygc as an alternative memory allocator for TinyGo WASI compilation; the default allocator may have performance issues with large data volumes ) -// Define a structure that needs to implement the sdk.Parser interface +// Define a struct that implements the sdk.Parser interface type plugin struct { } /* - This returns an array indicating where the agent needs to call the corresponding Export function of the plugin. Currently, there are 3 hook points: - HOOK_POINT_HTTP_REQ Indicates before the HTTP request parsing is completed and returned - HOOK_POINT_HTTP_RESP Indicates before the HTTP response parsing is completed and returned - HOOK_POINT_PAYLOAD_PARSE Indicates protocol judgment and parsing + The returned array indicates the hook points where the agent should invoke the plugin's corresponding Export function. Currently, there are 3 hook points: + HOOK_POINT_HTTP_REQ Before returning after HTTP request parsing is complete + HOOK_POINT_HTTP_RESP Before returning after HTTP response parsing is complete + HOOK_POINT_PAYLOAD_PARSE Protocol detection and parsing */ func (p plugin) HookIn() []sdk.HookBitmap { return []sdk.HookBitmap{ @@ -52,13 +57,13 @@ func (p plugin) HookIn() []sdk.HookBitmap { } } -// When HookIn() includes HOOK_POINT_HTTP_REQ, it will be called before the HTTP request parsing is completed and returned. -// HttpReqCtx contains BaseCtx and some parsed HTTP headers +// When HookIn() includes HOOK_POINT_HTTP_REQ, it will be called before returning after HTTP request parsing is complete. +// HttpReqCtx contains BaseCtx and some already parsed HTTP headers func (p plugin) OnHttpReq(ctx *sdk.HttpReqCtx) sdk.Action { - // baseCtx includes some information like IP, port, layer 4 protocol, packet direction, etc. + // baseCtx includes IP, port, L4 protocol, packet direction, etc. baseCtx := &ctx.BaseCtx - // Optional filtering by port and path + // Optional port and path filtering if baseCtx.DstPort != 8080 || !strings.HasPrefix(ctx.Path, "/user_info?") { return sdk.ActionNext() } @@ -83,24 +88,24 @@ func (p plugin) OnHttpReq(ctx *sdk.HttpReqCtx) sdk.Action { /* - When HookIn() includes HOOK_POINT_HTTP_RESP, it will be called before the HTTP response parsing is completed and returned. - HttpRespCtx contains BaseCtx and response code - The rest of the processing is basically the same as OnHttpReq + When HookIn() includes HOOK_POINT_HTTP_RESP, it will be called before returning after HTTP response parsing is complete. + HttpRespCtx contains BaseCtx and the response code. + The rest of the processing is basically the same as OnHttpReq. */ func (p plugin) OnHttpResp(ctx *sdk.HttpRespCtx) sdk.Action { return sdk.ActionNext() } /* - When HookIn() includes HOOK_POINT_PAYLOAD_PARSE, it will be called during protocol judgment - It needs to return a unique protocol number and protocol name, returning 0 indicates failure + When HookIn() includes HOOK_POINT_PAYLOAD_PARSE, it will be called during protocol detection. + You need to return a unique protocol number and protocol name; returning 0 as the protocol number indicates failure. */ func (p plugin) OnCheckPayload(baseCtx *sdk.ParseCtx) (uint8, string) { return 0, "" } func (p plugin) OnParsePayload(baseCtx *sdk.ParseCtx) sdk.ParseAction { - // ctx.L7 is the protocol number returned by OnCheckPayload, you can filter based on layer 4 protocol or protocol number first. + // ctx.L7 is the protocol number returned by OnCheckPayload; you can filter based on L4 protocol or protocol number first. if ctx.L4 != sdk.TCP || ctx.L7 != 1 { return sdk.ActionNext() } @@ -121,21 +126,21 @@ func (p plugin) OnParsePayload(baseCtx *sdk.ParseCtx) sdk.ParseAction { Req *Request Resp *Response Trace *Trace // Tracing information - Kv []KeyVal // Corresponding attribute + Kv []KeyVal // Corresponding attributes } type Request struct { - ReqType string // Corresponding request type - Domain string // Corresponding request domain - Resource string // Corresponding request resource - Endpoint string // Corresponding endpoint + ReqType string // Request type + Domain string // Request domain + Resource string // Request resource + Endpoint string // Endpoint } type Response struct { - Status RespStatus // Corresponding response status - Code *int32 // Corresponding response code - Result string // Corresponding response result - Exception string // Corresponding response exception + Status RespStatus // Response status + Code *int32 // Response code + Result string // Response result + Exception string // Response exception } */ return sdk.ParseActionAbortWithL7Info([]*sdk.L7ProtocolInfo{}) @@ -149,62 +154,62 @@ func main() { } // About return values /* - The agent can load multiple wasm plugins, and the agent will iterate through all plugins to call the corresponding Export functions, but the iteration behavior can be controlled by the return values + The agent can load multiple wasm plugins and will iterate through all plugins to call the corresponding Export functions, but the iteration behavior can be controlled via return values. - The return values are as follows: - sdk.ActionNext() Stop the current plugin and directly execute the next plugin - sdk.ActionAbort() Stop the current plugin and stop the iteration - sdk.ActionAbortWithErr(err) Stop the current plugin, print error logs, and stop the iteration + Return values include: + sdk.ActionNext() Stop current plugin and execute the next plugin + sdk.ActionAbort() Stop current plugin and stop iteration + sdk.ActionAbortWithErr(err) Stop current plugin, log the error, and stop iteration sdk.HttpActionAbortWithResult() - sdk.ParseActionAbortWithL7Info() The agent stops the iteration and extracts the corresponding return result + sdk.ParseActionAbortWithL7Info() Agent stops iteration and extracts the corresponding return result */ ``` -# Compiling and Loading Plugins +# Compiling and Loading the Plugin -## Compiling Plugins +## Compile the plugin -Use the following command to compile the Wasm program +Use the following command to compile the Wasm program: ```sh -# Using nottinygc to replace TinyGo's original memory allocator requires adding compilation parameters: -gc=custom and -tags=custommalloc +# To replace TinyGo's default memory allocator with nottinygc, add the compile parameters: -gc=custom and -tags=custommalloc tinygo build -o wasm.wasm -gc=custom -tags=custommalloc -target=wasi -panic=trap -scheduler=none -no-debug ./main.go ``` -## Uploading Plugins +## Upload the plugin -Upload the wasm file to the corresponding node and execute +Upload the wasm file to the corresponding node and execute: ```sh deepflow-ctl plugin create --type wasm --image wasm.wasm --name wasm ``` -## Loading Plugins +## Load the plugin -Add the following in the [agent-group configuration](../../best-practice/agent-advanced-config/#agent-group-config-常用操作) +Add the following in the [agent-group configuration](../../best-practice/agent-advanced-config/#agent-group-config-常用操作): ``` wasm_plugins: - - wasm // Corresponds to the name of the plugin uploaded by deepflow-ctl + - wasm // Corresponds to the name of the plugin uploaded via deepflow-ctl ``` # Related Issues and Limitations -- Cannot use go func(), you can remove the -scheduler=none parameter to pass the compilation but it will not achieve the desired effect -- Cannot use time.Sleep(), this will cause the Wasm plugin to fail to load -- If the plugin execution time is too long, it will block the agent's execution for a long time. If it enters an infinite loop, the agent will be continuously blocked -- Tinygo has certain limitations on Go's standard library and third-party libraries, not all Go code or libraries can be used. For the standard library, you can refer to [tinygo package supported](https://tinygo.org/docs/reference/lang-support/stdlib/) for the support status. Note that this list is for reference only, "Passes tests" showing "no" does not mean it cannot be used at all, for example, fmt.Sprintf() can be used but fmt.Println() cannot. -- Since Go version 1.21 supports WASI, if you need to use built-in serialization-related libraries (json, yaml, xml, etc.), you need to use Go version 1.21 or higher and Tinygo version 0.29 or higher. -- The structures returned from the Parser (L7ProtocolInfo, Trace, []KeyVal) will be serialized to linear memory. Currently, the memory allocated for serialization of each structure is fixed at 1 page (65536 bytes). If the returned structure is too large, serialization will fail. -- The agent determines the application layer protocol of a stream by iterating through all supported protocols. The current order is HTTP -> Wasm Hook -> DNS -> ..., with Wasm having a priority just below HTTP. Therefore, user-defined protocol judgment and parsing can override the agent's existing protocol judgment and parsing (except for HTTP/HTTP2). For example, in [this example](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/example/dns/dns.go), DNS parsing can be overridden, and the agent will not execute the default DNS parsing logic. -- Due to the complexity of network environments and protocols, incomplete application layer data frames may be received, such as IP fragmentation caused by MTU limitations, TCP peer receive window or flow control congestion window shrinkage, MSS being too small, etc., resulting in incomplete application layer data frames. Currently, transport layer connection tracking is not implemented. Additionally, application layer data that is too long will also be truncated. +- Cannot use `go func()`. You can remove the `-scheduler=none` parameter to compile, but it will not produce the desired effect. +- Cannot use `time.Sleep()`, as this will prevent the Wasm plugin from loading. +- If the plugin execution time is too long, it will block the agent for a long time; if it enters an infinite loop, the agent will remain blocked. +- tinygo has certain limitations for Go's standard library and third-party libraries; not all Go code or libraries can be used. For standard library support, refer to [tinygo package supported](https://tinygo.org/docs/reference/lang-support/stdlib/). Note that this list is for reference only; "Passes tests" showing "no" does not mean it cannot be used at all. For example, `fmt.Sprintf()` works, but `fmt.Println()` does not. +- Since Go 1.21 supports wasi, if you need to use built-in serialization-related libraries (json, yaml, xml, etc.), you need Go version >= 1.21 and tinygo version >= 0.29. +- Structures returned from Parser (L7ProtocolInfo, Trace, []KeyVal) are serialized into linear memory. Currently, each structure's serialization allocates a fixed 1 page (65536 bytes) of memory; if the returned structure is too large, serialization will fail. +- The agent determines a stream's application layer protocol by iterating through all supported protocols. The current order is HTTP -> Wasm Hook -> DNS -> ... . Wasm's priority is second only to HTTP, so user-defined protocol detection and parsing can override the agent's existing protocol detection and parsing (except HTTP/HTTP2). For example, in [this example](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/example/dns/dns.go), DNS parsing is overridden, and the agent will not execute the default DNS parsing logic. +- Due to network environment and protocol complexity, incomplete application layer data frames may be received (e.g., IP fragmentation due to MTU limits, TCP receive window or flow control congestion window shrinkage, small MSS, etc.), making it impossible to obtain complete application layer data frames. Transport layer connection tracking is not yet implemented. Additionally, overly long application layer data will be truncated. # Wasm Plugin Execution Flow -Before understanding the execution flow of the Wasm plugin, you need to have a general understanding of DeepFlow's protocol parsing. You can refer to [DeepFlow Protocol Development Documentation](https://github.com/deepflowio/deepflow/blob/main/docs/HOW_TO_SUPPORT_YOUR_PROTOCOL_CN.MD). +Before understanding the Wasm plugin execution flow, you should have a general understanding of deepflow's protocol parsing. You can refer to [DeepFlow Protocol Development Documentation](https://github.com/deepflowio/deepflow/blob/main/docs/HOW_TO_SUPPORT_YOUR_PROTOCOL_CN.MD). -The execution flow of the Wasm plugin is as follows +The Wasm plugin execution flow is as follows: ```mermaid graph TB; @@ -256,23 +261,23 @@ graph TB; id43(["host replace the trace info from struct Trace and extend attribute from []KeyVal"]); ``` -Among them, there are 6 structures for serialization/deserialization: +The serialized/deserialized structures include 6 types: - VmCtxBase - - Currently, when calling all Export functions, the host will serialize VmCtxBase to linear memory. The serialization format can be referenced [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/vm.rs#L199) - - Similarly, the instance will also deserialize it. The specific code can be referenced [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L73). + - In all current Export function calls, the host serializes VmCtxBase into linear memory. The serialization format can be found [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/vm.rs#L199) + - Similarly, the instance will deserialize it; see the code [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L73). - L7ProtocolInfo - - At the end of the Export function parse_payload, the instance will serialize L7ProtocolInfo to linear memory. The serialization format and code can be referenced [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L335) - - The host will also deserialize it. The code can be referenced [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/mod.rs#L152). + - At the end of the parse_payload Export function, the instance serializes L7ProtocolInfo into linear memory. The serialization format and code can be found [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L335) + - The host will also deserialize it; see the code [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/mod.rs#L152). - VmHttpReqCtx - - Before the HTTP request parsing is completed and returned, the Export function on_http_req will be called. The host will serialize VmCtxBase and VmHttpReqCtx to the instance's linear memory - - The serialization code and format of VmHttpReqCtx can be referenced [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/vm.rs#L328) - - The instance deserialization code can be referenced [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L173). + - Before returning after HTTP request parsing is complete, the Export function on_http_req is called, and the host serializes VmCtxBase and VmHttpReqCtx into the instance's linear memory. + - The serialization code and format for VmHttpReqCtx can be found [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/vm.rs#L328) + - The instance deserialization code can be found [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L173). - VmHttpRespCtx - - Before the HTTP response parsing is completed and returned, the Export function on_http_resp will be called. The host will serialize VmCtxBase and VmHttpRespCtx to the instance's linear memory - - The serialization format of VmHttpRespCtx can be referenced [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/vm.rs#L395) - - The instance deserialization code can be referenced [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L232). + - Before returning after HTTP response parsing is complete, the Export function on_http_resp is called, and the host serializes VmCtxBase and VmHttpRespCtx into the instance's linear memory. + - The serialization format for VmHttpRespCtx can be found [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/vm.rs#L395) + - The instance deserialization code can be found [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L232). - Trace, []KeyVal - - Before the Export function on_http_req/on_http_resp returns, the instance will serialize Trace and []KeyVal to linear memory - - The serialization code can be referenced [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L515) - - The deserialization code and format can be referenced [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/abi_import.rs#L376). \ No newline at end of file + - Before returning from the Export functions on_http_req/on_http_resp, the instance serializes Trace and []KeyVal into linear memory. + - The serialization code can be found [here](https://github.com/deepflowio/deepflow-wasm-go-sdk/blob/5393818adf94f2f9b296de82e20f614ba3b2336a/sdk/serde.go#L515) + - The deserialization code and format can be found [here](https://github.com/deepflowio/deepflow/blob/0da738106f710cad9bbce6632384105b1b868e59/agent/src/plugin/wasm/abi_import.rs#L376). \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/01-metrics/01-metrics-auto-tagging.md b/translate/translated/08-integration/02-input/01-metrics/01-metrics-auto-tagging.md index da56e596..3146dce9 100644 --- a/translate/translated/08-integration/02-input/01-metrics/01-metrics-auto-tagging.md +++ b/translate/translated/08-integration/02-input/01-metrics/01-metrics-auto-tagging.md @@ -5,6 +5,6 @@ permalink: /integration/input/metrics/metrics-auto-tagging > This document was translated by ChatGPT -DeepFlow automatically calculates all related [cloud resources, K8s resources, K8s Label/Annotation/Env](../../../features/auto-tagging/eliminate-data-silos/) tags based on existing tags such as `pod_name` (Telegraf), `pod` (Prometheus), `instance` (Prometheus) in the metric data. This enables the integration of metric data with other observability data and enhances the drill-down capability of integrated data. +DeepFlow automatically calculates all related [cloud resources, K8s resources, K8s Label/Annotation/Env](../../../features/auto-tagging/eliminate-data-silos/) tags based on existing labels in metric data such as `pod_name` (Telegraf), `pod` (Prometheus), and `instance` (Prometheus). This enables the integrated metric data to be seamlessly connected with other observability data, enhancing the drill-down capabilities of the integrated data. -With the AutoTagging capability, application developers no longer need to worry about inserting a large number of tags into the metric data. All tag injections will be automatically completed with the activation of business resources, microservice releases, and other traffic. Additionally, DeepFlow's [SmartEncoding](../../../features/auto-tagging/smart-encoding/) mechanism ensures that the automatically inserted tags consume minimal resource overhead. Finally, for the large number of tags already injected in Prometheus and Telegraf, thanks to ClickHouse's sparse index mechanism, developers no longer need to worry about high cardinality issues. +With the AutoTagging capability, application developers no longer need to worry about inserting a large number of tags into metric data. All tag injection is automatically completed along with business resource provisioning, microservice releases, and other traffic events. In addition, DeepFlow’s [SmartEncoding](../../../features/auto-tagging/smart-encoding/) mechanism ensures that automatically inserted tags consume only minimal resources. Finally, for the large number of tags already injected in Prometheus and Telegraf, thanks to ClickHouse’s sparse index mechanism, developers no longer need to worry about high cardinality issues. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/01-metrics/02-prometheus.md b/translate/translated/08-integration/02-input/01-metrics/02-prometheus.md index eeade80c..e0fb49b8 100644 --- a/translate/translated/08-integration/02-input/01-metrics/02-prometheus.md +++ b/translate/translated/08-integration/02-input/01-metrics/02-prometheus.md @@ -1,5 +1,5 @@ --- -title: Integrate Prometheus Data +title: Integrating Prometheus Data permalink: /integration/input/metrics/prometheus --- @@ -24,8 +24,8 @@ end ## Install Prometheus -You can learn the relevant background knowledge in the [Prometheus documentation](https://prometheus.io/docs/introduction/overview/). -If you do not have Prometheus in your cluster, you can quickly deploy a Prometheus in the `deepflow-prometheus-demo` namespace using the following steps: +You can learn the relevant background information in the [Prometheus documentation](https://prometheus.io/docs/introduction/overview/). +If your cluster does not have Prometheus, you can quickly deploy one in the `deepflow-prometheus-demo` namespace using the following steps: ```bash # add helm chart @@ -40,15 +40,15 @@ helm install prometheus prometheus-community/prometheus -n deepflow-prometheus-d We need to configure Prometheus `remote_write` to send data to the DeepFlow Agent. -First, determine the address of the data listening service started by the DeepFlow Agent. After [installing the DeepFlow Agent](../../../ce-install/single-k8s/), the DeepFlow Agent Service address will be displayed. Its default value is `deepflow-agent.default`. Please fill in the actual service name and namespace in the configuration. +First, determine the address of the data listening service started by the DeepFlow Agent. After [installing DeepFlow Agent](../../../ce-install/single-k8s/), the DeepFlow Agent Service address will be displayed. Its default value is `deepflow-agent.default`. Please fill in the actual service name and namespace in the configuration. -Execute the following command to modify the default configuration of Prometheus (assuming it is in `deepflow-prometheus-demo`): +Run the following command to modify the default Prometheus configuration (assuming it is in `deepflow-prometheus-demo`): ```bash kubectl edit cm -n deepflow-prometheus-demo prometheus-server ``` -Configure the `remote_write` address (please change `DEEPFLOW_AGENT_SVC` to the service name of deepflow-agent): +Configure the `remote_write` address (replace `DEEPFLOW_AGENT_SVC` with the service name of deepflow-agent): ```yaml remote_write: @@ -57,7 +57,7 @@ remote_write: ## Configure remote_read -If you want Prometheus to query data from DeepFlow, you need to configure Prometheus `remote_read` (please change `DEEPFLOW_SERVER_SVC` to the service name of deepflow-server): +If you want Prometheus to query data from DeepFlow, you need to configure Prometheus `remote_read` (replace `DEEPFLOW_SERVER_SVC` with the service name of deepflow-server): ```yaml remote_read: @@ -65,28 +65,76 @@ remote_read: read_recent: true ``` -# Configure DeepFlow (Deprecated in v6.5 and later versions) +# Configure DeepFlow (deprecated in v6.5 and later) -Please refer to the section [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) and add the configuration for the `prometheus targets api` address (not required for versions prior to v6.2) to complete the DeepFlow Agent configuration. The purpose is to synchronize prometheus activeTargets.labels and config to deepflow-server to improve storage and query performance. +Refer to the section [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) and add the `prometheus targets api` address configuration (not required for v6.2 and earlier) to complete the DeepFlow Agent configuration. +The purpose is to synchronize prometheus activeTargets.labels and config to deepflow-server to improve storage and query performance. -Add the following configuration for the Group where the Agent is located (please modify `PROMETHEUS_HTTP_API_ADDRESSES`): +Add the following configuration to the Group where the Agent resides (replace `PROMETHEUS_HTTP_API_ADDRESSES`): ```yaml prometheus_http_api_addresses: # Required when integrating Prometheus metrics - { PROMETHEUS_HTTP_API_ADDRESSES } ``` -# View Prometheus Data +# Integrating xExporter Data -The metrics in Prometheus will be stored in the `prometheus` database of DeepFlow. -The original labels of Prometheus can be referenced through tag.XXX, and the metric values can be referenced through value. -At the same time, DeepFlow will automatically inject a large number of Meta Tags and Custom Tags, allowing the data collected by Prometheus to be seamlessly associated with other data sources. +In the enterprise edition of DeepFlow-Agent, it supports directly pulling metrics from any xExporter compatible with the Prometheus ecosystem and pushing them into DeepFlow. -Using Grafana, select the `DeepFlow` data source to display the search results as shown below: +## Data Flow + +```mermaid +flowchart TD + +subgraph K8s-Cluster + xExporter["x-Exporter (deployment)"] + DeepFlowAgent["deepflow-agent-ee (daemonset)"] + DeepFlowServer["deepflow-server (deployment)"] + + xExporter -->|metrics| DeepFlowAgent + DeepFlowAgent -->|metrics| DeepFlowServer +end +``` + +## Configure DeepFlow (available in DeepFlow-Agent v6.6 and later) + +Add the following configuration to the Group where the Agent resides: + +```yaml +inputs: + vector: + config: + sources: + x_exporter: + type: prometheus_scrape + endpoints: + - http://${HOST:PORT}/metrics + scrape_interval_secs: 10 + scrape_timeout_secs: 10 + honor_labels: true + instance_tag: instance + endpoint_tag: metrics_endpoint + sinks: + prometheus_remote_write: + type: prometheus_remote_write + inputs: + - x_exporter + endpoint: http://127.0.0.1:38086/api/v1/prometheus + healthcheck: + enabled: false +``` + +# Viewing Prometheus Data + +Metrics from Prometheus will be stored in the `prometheus` database of DeepFlow. +Original Prometheus labels can be referenced via `tag.XXX`, and metric values via `value`. +At the same time, DeepFlow will automatically inject a large number of Meta Tags and Custom Tags, enabling Prometheus-collected data to be seamlessly correlated with other data sources. + +When using Grafana and selecting the `DeepFlow` data source for search, the display is as shown below: ![Prometheus Data Integration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20231003651c19e6684d1.png) # Notes 1. When calculating the `Derivative` operator through the DeepFlow data source, you must select an outer operator (such as `Avg`). -2. When calculating the `Derivative` operator, since the calculation process first calculates the `Derivative` operator for multiple time series of the same metric, and then calculates the outer operator, if the query interval is less than the data collection interval, using relative time queries like now-xx, if multiple time series of the same metric are continuously writing new data, the results of multiple queries may be inconsistent. \ No newline at end of file +2. When calculating the `Derivative` operator, since the calculation process first applies the `Derivative` operator to multiple time series of the same metric and then applies the outer operator, if the query interval is shorter than the data collection interval, using relative time queries such as now-xx, and multiple time series of the same metric are continuously writing new data, the results of multiple queries may be inconsistent. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/01-metrics/03-telegraf.md b/translate/translated/08-integration/02-input/01-metrics/03-telegraf.md index 40ced6bd..f1ff2ccb 100644 --- a/translate/translated/08-integration/02-input/01-metrics/03-telegraf.md +++ b/translate/translated/08-integration/02-input/01-metrics/03-telegraf.md @@ -28,12 +28,12 @@ subgraph Host end ``` -# Configuring Telegraf +# Configure Telegraf -## Installing Telegraf +## Install Telegraf -You can learn the relevant background knowledge in the [Telegraf documentation](https://www.influxdata.com/time-series-platform/telegraf/). -If your cluster does not have Telegraf, you can quickly deploy Telegraf as a DaemonSet using the following steps: +Refer to the [Telegraf documentation](https://www.influxdata.com/time-series-platform/telegraf/) for background information. +If Telegraf is not installed in your cluster, you can quickly deploy it as a DaemonSet using the following steps: ```bash # add helm chart @@ -46,21 +46,22 @@ helm upgrade --install telegraf influxdata/telegraf -n deepflow-telegraf-demo -- kubectl apply -f https://raw.githubusercontent.com/deepflowio/deepflow-demo/main/DeepFlow-Telegraf-Demo/deepflow-telegraf-demo.yaml ``` -## Configuring Telegraf Data Output +## Configure Telegraf Data Output -We need to modify Telegraf's configuration to send data to the DeepFlow Agent. +We need to modify Telegraf’s configuration so that it sends data to the DeepFlow Agent. -First, we need to determine the address of the data listening service started by the DeepFlow Agent. After [installing the DeepFlow Agent](../../../ce-install/single-k8s/), -the DeepFlow Agent Service address will be displayed, with a default value of `deepflow-agent.default`. -If you have modified it, please fill in the actual service name and namespace in the configuration. +First, determine the address of the data listening service started by the DeepFlow Agent. +After [installing the DeepFlow Agent](../../../ce-install/single-k8s/), +the DeepFlow Agent Service address will be displayed, with the default value being `deepflow-agent.default`. +If you have changed it, update the configuration with the actual service name and namespace. -Next, modify Telegraf's default configuration (assuming it is in the `deepflow-telegraf-demo` namespace): +Next, modify Telegraf’s default configuration (assuming it is in the `deepflow-telegraf-demo` namespace): ```bash kubectl edit cm -n deepflow-telegraf-demo telegraf ``` -In `telegraf.conf`, add the following configuration (please change `DEEPFLOW_AGENT_SVC` to the service name of deepflow-agent): +In `telegraf.conf`, add the following configuration (replace `DEEPFLOW_AGENT_SVC` with the service name of deepflow-agent): ```toml [[outputs.http]] @@ -68,18 +69,19 @@ In `telegraf.conf`, add the following configuration (please change `DEEPFLOW_AGE data_format = "influx" ``` -# Configuring DeepFlow +# Configure DeepFlow -Please refer to the section [Configuring DeepFlow](../tracing/opentelemetry/#配置-deepflow) to complete the DeepFlow Agent configuration. +Refer to the section [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) to complete the DeepFlow Agent configuration. -# Viewing Telegraf Data +# View Telegraf Data -Metrics from Telegraf will be stored in DeepFlow's `ext_metrics` database. -To reduce the number of tables, DeepFlow will store all Measurements in a single ClickHouse Table, -but users will still see a series of data tables corresponding to the original Telegraf Measurements. -The original tags of Telegraf metrics can be referenced via tag.XXX, and metric values can be referenced via metrics.YYY. -At the same time, DeepFlow will automatically inject a large number of Meta Tags and Custom Tags, allowing Telegraf-collected data to be seamlessly associated with other data sources. +Metrics from Telegraf will be stored in DeepFlow’s `ext_metrics` database. +To reduce the number of tables, DeepFlow stores all Measurements in a single ClickHouse table. +When queried, users will still see a series of tables corresponding to Telegraf’s original Measurements. +The original tags of Telegraf metrics can be referenced via `tag.XXX`, and metric values via `metrics.YYY`. +At the same time, DeepFlow automatically injects a large number of Meta Tags and Custom Tags, +allowing Telegraf-collected data to be seamlessly correlated with other data sources. -Using Grafana, select the `DeepFlow` data source to display the search results as shown below: +When using Grafana, selecting the `DeepFlow` data source for queries will display results as shown below: ![Telegraf Data Integration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20231003651c1adb93461.png) \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/01-metrics/04-grafana-agent.md b/translate/translated/08-integration/02-input/01-metrics/04-grafana-agent.md index 58a4e0ac..8a9c613b 100644 --- a/translate/translated/08-integration/02-input/01-metrics/04-grafana-agent.md +++ b/translate/translated/08-integration/02-input/01-metrics/04-grafana-agent.md @@ -1,5 +1,5 @@ --- -title: Integrate Grafana Agent Data +title: Integrating Grafana Agent Data permalink: /integration/input/metrics/grafana-agent --- @@ -32,8 +32,8 @@ end ## Grafana Agent -You can learn the relevant background knowledge in the [Grafana Agent Documentation](https://grafana.com/docs/agent/latest/). -If your environment does not have Grafana Agent, you can deploy it using the following steps: +Refer to the [Grafana Agent documentation](https://grafana.com/docs/agent/latest/) for background information. +If Grafana Agent is not available in your environment, you can deploy it using the following steps: ::: code-tabs#shell @@ -136,6 +136,8 @@ helm repo update cat << EOF > grafana-agent-values-custom.yaml agent: + # -- Address to listen for traffic on. 0.0.0.0 exposes the UI to other + # containers. # -- Address to listen for traffic on. 0.0.0.0 exposes the UI to other # containers. listenAddr: \$(HOSTIP) @@ -504,10 +506,12 @@ helm install grafana-agent grafana/grafana-agent \ # Configure DeepFlow -Please refer to the section [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) to complete the configuration of the DeepFlow Agent. +Please refer to the section [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) to complete the DeepFlow Agent configuration. -Metrics from the Grafana Agent will be stored in the `Grafana Agent` database within DeepFlow. The original tags from Grafana Agent can be referenced using `tag.XXX`, and metric values can be referenced using `value`. Additionally, DeepFlow will automatically inject a large number of Meta Tags and Custom Tags, allowing the data collected by Grafana Agent to seamlessly correlate with other data sources. +Metrics from Grafana Agent will be stored in DeepFlow’s `Grafana Agent` database. +The original labels from Grafana Agent can be referenced via `tag.XXX`, and metric values can be referenced via `value`. +At the same time, DeepFlow will automatically inject a large number of Meta Tags and Custom Tags, enabling Grafana Agent’s collected data to be seamlessly correlated with other data sources. -When using Grafana and selecting the `DeepFlow` data source for search, the display will appear as shown below: +When using Grafana and selecting the `DeepFlow` data source for queries, the display will look like the following: ![Grafana Agent Data Integration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20231003651c19e6684d1.png) diff --git a/translate/translated/08-integration/02-input/02-tracing/01-full-stack-distributed-tracing.md b/translate/translated/08-integration/02-input/02-tracing/01-full-stack-distributed-tracing.md index 4e5707a0..38f82f5b 100644 --- a/translate/translated/08-integration/02-input/02-tracing/01-full-stack-distributed-tracing.md +++ b/translate/translated/08-integration/02-input/02-tracing/01-full-stack-distributed-tracing.md @@ -5,23 +5,23 @@ permalink: /integration/input/tracing/full-stack-distributed-tracing > This document was translated by ChatGPT -DeepFlow leverages eBPF to innovatively implement AutoTracing, covering distributed tracing at both the API call and network transmission levels. This complements OpenTelemetry's function-level coverage within application code. +DeepFlow leverages eBPF innovations to implement AutoTracing, covering distributed tracing at both the API call and network transmission layers. This complements OpenTelemetry’s coverage at the function granularity within application code. -By integrating APM data sources such as OpenTelemetry and SkyWalking, AutoTracing capabilities are further enhanced, enabling full-stack distributed tracing across applications, systems, and networks. In the flame graph below, we can observe: +By integrating APM data sources such as OpenTelemetry and SkyWalking, the AutoTracing capability becomes more complete, enabling full-stack distributed tracing across applications, systems, and networks. In the flame graph below, we can see: -- Complex and lengthy gateway paths can be traced, including API gateways, microservice gateways, load balancers, Ingress, etc. -- Upstream and downstream calls of any microservice can be traced, including often-overlooked calls like DNS, and services that cannot be instrumented like MySQL and Redis. -- Full-stack network paths between any two microservices can be traced, covering from application code to system calls, container network components like Sidecar/iptables/ipvs, virtual machine network components like OvS/LinuxBridge, and cloud network components like NFV gateways. +- The ability to trace complex and lengthy gateway paths, including API gateways, microservice gateways, load balancers, Ingress, and more +- The ability to trace upstream and downstream calls of any microservice, including calls often overlooked by developers such as DNS, as well as services like MySQL and Redis that cannot be instrumented +- The ability to trace the full-stack network path between any two microservices, covering everything from application code to system calls, container network components such as Sidecar/iptables/ipvs, virtual machine network components such as OvS/LinuxBridge, and cloud network components such as NFV gateways -![DeepFlow and APM Integration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20231002651a886330ed3.png) +![DeepFlow with APM Integration](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20231002651a886330ed3.png) -There are several methods for integrating DeepFlow with APM, categorized into four types: +There are several ways to integrate DeepFlow with APM, categorized into the following four types: -- Full-stack distributed tracing results displayed by DeepFlow - - 1. **API Call**: DeepFlow calls the APM's `Trace API` to obtain APP Spans from the APM. Configuration methods can be found in [Calling APM Trace API](./apm-trace-api/) - - 2. **Data Import**: DeepFlow receives APP Spans exported by APM via the OTLP protocol. Configuration methods can be found in [Importing OpenTelemetry Data](./opentelemetry/) and [Importing SkyWalking Data](./skywalking/) -- Full-stack distributed tracing results displayed by APM - - 3. **Providing API**: APM calls the `Trace Completion API` provided by DeepFlow to obtain SYS Spans and NET Spans from DeepFlow. Configuration methods can be found in the [Trace Completion API](../../output/query/trace-completion/) documentation - - 4. **Data Export**: APM receives SYS Spans and NET Spans exported by DeepFlow via the OTLP protocol. Configuration methods can be found in the [OTLP Exporter](../../output/export/opentelemetry-exporter/) documentation +- Full-stack distributed tracing results displayed by DeepFlow + - 1. **Call API**: DeepFlow calls the APM’s `Trace API` to obtain APP Spans from the APM. For configuration, refer to [Calling APM Trace API](./apm-trace-api/) + - 2. **Import Data**: DeepFlow receives APP Spans exported by the APM via the OTLP protocol. For configuration, refer to [Importing OpenTelemetry Data](./opentelemetry/) and [Importing SkyWalking Data](./skywalking/) +- Full-stack distributed tracing results displayed by the APM + - 3. **Provide API**: The APM calls the `Trace Completion API` provided by DeepFlow to obtain SYS Spans and NET Spans from DeepFlow. For configuration, refer to the [Trace Completion API](../../output/query/trace-completion/) documentation + - 4. **Export Data**: The APM receives SYS Spans and NET Spans exported by DeepFlow via the OTLP protocol. For configuration, refer to the [OTLP Exporter](../../output/export/opentelemetry-exporter/) documentation -Among the four methods mentioned above, methods 1 and 2 require the least effort, as they only need configuration; method 3 also has a very small development workload, making it very suitable for initial use; method 4 requires understanding the association logic between SYS Span, NET Span, and APP Span, and has a larger development workload, making it suitable for use after gaining a deep understanding of DeepFlow. \ No newline at end of file +Among these four methods, options 1 and 2 require the least effort and only need configuration; option 3 also requires minimal development work and is well-suited for early-stage use; option 4 requires understanding the correlation logic between SYS Spans, NET Spans, and APP Spans, involves more development work, and is better suited for use after gaining a deeper understanding of DeepFlow. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/02-tracing/02-traccing-auto-tagging.md b/translate/translated/08-integration/02-input/02-tracing/02-traccing-auto-tagging.md index 82a5a8ee..12f5652f 100644 --- a/translate/translated/08-integration/02-input/02-tracing/02-traccing-auto-tagging.md +++ b/translate/translated/08-integration/02-input/02-tracing/02-traccing-auto-tagging.md @@ -5,6 +5,6 @@ permalink: /integration/input/tracing/traccing-auto-tagging > This document was translated by ChatGPT -DeepFlow Agent needs to inject VPC and IP tags into the data so that DeepFlow Server can expand all other tags based on this. For tracing data, we hope that the OpenTelemetry Agent can tag IP information for the DeepFlow Agent. Fortunately, the [k8s attributes processor](https://pkg.go.dev/github.com/open-telemetry/opentelemetry-collector-contrib/processor/k8sattributesprocessor#section-readme) plugin can automatically inject the IP address of the Span sender, and `attribute.net.peer.ip` can be automatically injected by most OpenTelemetry Instrumentation. +The DeepFlow Agent needs to inject VPC and IP tags into the data so that the DeepFlow Server can extend all other tags based on them. For tracing data, we want the OpenTelemetry Agent to label IP information for the DeepFlow Agent. Fortunately, the [k8s attributes processor](https://pkg.go.dev/github.com/open-telemetry/opentelemetry-collector-contrib/processor/k8sattributesprocessor#section-readme) plugin can automatically inject the IP address of the Span sender, and `attribute.net.peer.ip` can be automatically injected by most OpenTelemetry Instrumentations. -Based on DeepFlow's AutoTagging capability, we have automated the association of tracing data with other observability data, without requiring developers to insert any code. +With DeepFlow’s AutoTagging capability, we can automatically associate tracing data with other observability data, without requiring developers to insert any code. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/02-tracing/03-apm-trace-api.md b/translate/translated/08-integration/02-input/02-tracing/03-apm-trace-api.md index 2e2f0f07..dc42a2ec 100644 --- a/translate/translated/08-integration/02-input/02-tracing/03-apm-trace-api.md +++ b/translate/translated/08-integration/02-input/02-tracing/03-apm-trace-api.md @@ -1,5 +1,5 @@ --- -title: Calling APM's Trace API +title: Calling the APM Trace API permalink: /integration/input/tracing/apm-trace-api --- @@ -7,7 +7,7 @@ permalink: /integration/input/tracing/apm-trace-api # Introduction -DeepFlow has the capability to obtain APP Spans from external APMs and associate these APP Spans with the tracing data collected by DeepFlow. Currently, only SkyWalking is supported as the external APM storage. Applications do not need any modifications; you only need to change the DeepFlow configuration to achieve DeepFlow's full-link, zero-instrumentation tracing capability. +DeepFlow has the capability to retrieve APP Spans from external APMs and associate them with tracing data collected by DeepFlow. Currently, only SkyWalking is supported as an external APM storage. Applications require no modifications—simply update the DeepFlow configuration to gain DeepFlow’s full-link, zero-instrumentation tracing capability. # Data Flow @@ -41,19 +41,43 @@ end # Configuration -Modify the [configuration](https://github.com/deepflowio/deepflow/blob/main/server/server.yaml) of the DeepFlow Server by adding the following content: +Modify the [configuration](https://github.com/deepflowio/deepflow/blob/main/server/server.yaml) of the DeepFlow Server and add the following content: ```yaml querier: external-apm: - name: skywalking - addr: 127.0.0.1:12800 # FIXME: Replace this with the address of the SkyWalking OAP Server, port 12800 is the default port for HTTP service + addr: 127.0.0.1:12800 # FIXME: Replace with the address of the SkyWalking OAP Server; port 12800 is the default HTTP service port ``` -At the same time, you need to modify the [configuration](https://github.com/deepflowio/deepflow-app/blob/main/app/app.yaml) of the DeepFlow App by setting the following value to `true`: +At the same time, modify the [configuration](https://github.com/deepflowio/deepflow-app/blob/main/app/app.yaml) of the DeepFlow App and set the following value to `true`: ```yaml app: spec: call_apm_api_to_supplement_trace: true -``` \ No newline at end of file +``` + +The mapping from SkyWalking APM’s [skywalking-query-protocol](https://github.com/apache/skywalking-query-protocol/blob/master/trace.graphqls) to DeepFlow flame graph is as follows: + +| Name | Chinese | SkyWalking Data Structure | Description | +| --------------- | ---------------- | --------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- | +| startTimeUs | 开始时间 | span.startTime | -- | +| endTimeUs | 结束时间 | span.endTime | -- | +| tapSide | 观测点 | span.spanType.Exit: Client App (C-APP), span.spanType.Entry: Server App (S-APP), span.spanType.Local: App (APP) | Observation point, i.e., observation_point, converted to the corresponding enum value | +| traceID | TraceID | span.traceID | -- | +| spanID | SpanID | span.segmentID-span.spanID | -- | +| parentSpanID | ParentSpanID | span.segmentID-span.parentSpanID/span.ref.parentSegmentID/span.ref.parentSpanID | When span.parentSpanID = -1, attempt to get span.ref as the parentSpan | +| spanKind | span 类型 | span.type=Exit: SPAN_KIND_CLIENT, span.type=Entry: SPAN_KIND_SERVER, span.type=Local: SPAN_KIND_INTERNAL | Converted to the corresponding enum value | +| endpoint | 请求端点 | span.endpointName | Specific resource requested; in HTTP protocol, usually the request route | +| appService | 应用服务 | span.serviceCode | -- | +| appInstance | 应用服务实例 | span.serviceInstance | -- | +| serviceUname | 服务名称 | span.serviceCode | -- | +| requestType | 请求类型 | span.tags.[http.method/cache.cmd/db.operation/rpc.method] | Get the value from tag according to the protocol | +| requestResource | 请求资源 | span.endpointName/span.tags.[db.statement/cache.key/url/http.url] | When span.endpointName does not exist, try to extract from http.url, only keeping the request info after the domain | +| responseCode | 响应码 | span.tags.[http.status.code/http.status_code/http.status] | -- | +| reponseStatus | 响应状态 | -- | Converted from responseCode: 2xx~3xx: STATUS_OK, 4xx: STATUS_CLIENT_ERROR, 5xx: STATUS_SERVER_ERROR | +| signalSource | 信号源 | OTEL(4) | Fixed enum value | +| l7Protocol | 应用协议 | span.layer=Http: HTTP, span.tags.[db.type/db.system/http.scheme/rpc.system/messaging.system/messaging.protocol] | If span.layer is not Http, try to get from span.tags; if any tag starts with `http.`, treat as HTTP | +| l7ProtocolStr | 应用协议(名称) | -- | Get the specific name according to the enum value of l7Protocol | +| attribute | 标签 | span.tags | -- | \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/02-tracing/04-opentelemetry.md b/translate/translated/08-integration/02-input/02-tracing/04-opentelemetry.md index 81568c4a..f598ee78 100644 --- a/translate/translated/08-integration/02-input/02-tracing/04-opentelemetry.md +++ b/translate/translated/08-integration/02-input/02-tracing/04-opentelemetry.md @@ -7,7 +7,7 @@ permalink: /integration/input/tracing/opentelemetry # Data Flow -Sending via otel-collector to deepflow-agent: +Send via otel-collector to deepflow-agent: ```mermaid flowchart TD @@ -38,7 +38,7 @@ subgraph Host end ``` -Sending directly to deepflow-agent: +Send directly to deepflow-agent: ```mermaid flowchart TD @@ -67,17 +67,19 @@ end # Configure OpenTelemetry -We recommend using the agent mode of otel-collector to send trace data to deepflow-agent to avoid data transmission across K8s nodes. Of course, using the gateway mode of otel-collector is also completely feasible. The following document introduces the deployment and configuration methods using otel-agent as an example. +We recommend using otel-collector in agent mode to send trace data to deepflow-agent to avoid cross-node data transfer in K8s. +Of course, using otel-collector in gateway mode is also fully supported. The following documentation uses otel-agent as an example to describe deployment and configuration. ## Install otel-agent -Refer to the [OpenTelemetry documentation](https://opentelemetry.io/docs/) for relevant background knowledge. If OpenTelemetry is not yet available in your environment, you can quickly deploy an otel-agent DaemonSet in the `open-telemetry` namespace using the following command: +Refer to the [OpenTelemetry documentation](https://opentelemetry.io/docs/) for background information. +If OpenTelemetry is not yet installed in your environment, you can quickly deploy an otel-agent DaemonSet in the `open-telemetry` namespace with the following command: ```bash kubectl apply -n open-telemetry -f https://raw.githubusercontent.com/deepflowio/deepflow-demo/main/open-telemetry/open-telemetry.yaml ``` -After installation, you can see a list of components in the environment: +After installation, you should see the following components in your environment: ```bash kubectl get all -n open-telemetry @@ -89,7 +91,8 @@ kubectl get all -n open-telemetry | Service | otel-agent | | ConfigMap | otel-agent | -If you need to use other versions or updated opentelemetry-collector-contrib, find the desired image version in the [otel-docker](https://hub.docker.com/r/otel/opentelemetry-collector-contrib/tags) repository, and then update the image using the following command: +If you need another version or a newer opentelemetry-collector-contrib, +visit the [otel-docker](https://hub.docker.com/r/otel/opentelemetry-collector-contrib/tags) repository to find the desired image version, then update the image with: ```bash LATEST_TAG="xxx" # FIXME @@ -99,14 +102,14 @@ kubectl set image -n open-telemetry daemonset/otel-agent otel-agent=otel/opentel ## Configure otel-agent -We need to configure `otel-agent-config.exporters.otlphttp` in the otel-agent ConfigMap to send traces to DeepFlow. First, query the current configuration: +We need to configure `otel-agent-config.exporters.otlphttp` in the otel-agent ConfigMap to send traces to DeepFlow. First, check the current configuration: ```bash kubectl get cm -n open-telemetry otel-agent-conf -o custom-columns=DATA:.data | \ grep -A 5 otlphttp: ``` -deepflow-agent uses ClusterIP Service to receive traces, modify the otel-agent configuration as follows: +deepflow-agent uses a ClusterIP Service to receive traces, so modify the otel-agent configuration as follows: ```yaml otlphttp: @@ -117,7 +120,7 @@ otlphttp: enabled: true ``` -Additionally, to ensure the IP on the Span sending side is passed to DeepFlow, add the following configuration: +Additionally, to ensure the sender's IP of the Span is passed to DeepFlow, add the following configuration: ```yaml processors: @@ -129,13 +132,13 @@ processors: action: insert ``` -Finally, in the service.pipeline, add the following to the `traces` section: +Finally, in the `service.pipeline` section, add the following under `traces`: ```yaml service: pipelines: traces: - processors: [k8sattributes, resource] # Ensure k8sattributes processor is processed first + processors: [k8sattributes, resource] # Ensure k8sattributes processor runs first exporters: [otlphttp] ``` @@ -143,25 +146,25 @@ service: Next, we need to enable the data receiving service of deepflow-agent. -First, determine the collector group ID where deepflow-agent is located, usually the ID of the group named default: +First, determine the collector group ID where deepflow-agent resides, usually the ID of the group named `default`: ```bash deepflow-ctl agent-group list ``` -Check if the collector group already has a configuration: +Check whether this collector group already has a configuration: ```bash deepflow-ctl agent-group-config list ``` -If there is already a configuration, export it to a yaml file for modification: +If a configuration exists, export it to a yaml file for modification: ```bash deepflow-ctl agent-group-config list -o yaml > your-agent-group-config.yaml ``` -Modify the yaml file to ensure it contains the following configuration items: +Edit the yaml file to ensure it contains the following: ```bash vtap_group_id: @@ -169,13 +172,13 @@ external_agent_http_proxy_enabled: 1 # required external_agent_http_proxy_port: 38086 # optional, default 38086 ``` -Update the collector group's configuration: +Update the collector group configuration: ``` deepflow-ctl agent-group-config update -f your-agent-group-config.yaml ``` -If the collector group does not yet have a configuration, create a new configuration based on the your-agent-group-config.yaml file using the following command: +If the collector group has no configuration, create one based on your-agent-group-config.yaml: ```bash deepflow-ctl agent-group-config create -f your-agent-group-config.yaml @@ -183,13 +186,14 @@ deepflow-ctl agent-group-config create -f your-agent-group-config.yaml # Experience with Spring Boot Demo -## Deploy Demo +## Deploy the Demo -This Demo is from [this GitHub repository](https://github.com/liuzhibin-cn/my-demo), which is a Spring Boot-based WebShop application composed of five microservices. Its architecture is as follows: +This demo comes from [this GitHub repository](https://github.com/liuzhibin-cn/my-demo). +It is a Spring Boot-based WebShop application composed of five microservices, with the following architecture: ![Sping Boot Demo Architecture](./imgs/spring-boot-webshop-arch.png) -Deploy this Demo with one command: +Deploy the demo with one command: ```bash kubectl apply -n deepflow-otel-spring-demo -f https://raw.githubusercontent.com/deepflowio/deepflow-demo/main/DeepFlow-Otel-Spring-Demo/deepflow-otel-spring-demo.yaml @@ -197,26 +201,29 @@ kubectl apply -n deepflow-otel-spring-demo -f https://raw.githubusercontent.com/ ## View Tracing Data -Go to Grafana, open the `Distributed Tracing` Dashboard, select `namespace = deepflow-otel-spring-demo`, and then choose a call to trace. DeepFlow can correlate tracing data obtained from OpenTelemetry, eBPF, and BPF in a single Trace flame graph, covering the full-stack call path of a Spring Boot application from business code, system functions, to network interfaces, achieving true end-to-end distributed tracing, as shown below: +Go to Grafana, open the `Distributed Tracing` Dashboard, select `namespace = deepflow-otel-spring-demo`, and choose a request to trace. +DeepFlow can correlate tracing data from OpenTelemetry, eBPF, and BPF into a single trace flame graph, +covering the full-stack call path of a Spring Boot application from business code, system functions, to network interfaces, achieving true end-to-end distributed tracing, as shown below: ![OTel Spring Demo](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2022082363044b24c3b37.png) -You can also visit [DeepFlow Online Demo](https://ce-demo.deepflow.yunshan.net/d/Distributed_Tracing/distributed-tracing?var-namespace=deepflow-otel-spring-demo&from=deepflow-doc) to see the effect. +You can also visit the [DeepFlow Online Demo](https://ce-demo.deepflow.yunshan.net/d/Distributed_Tracing/distributed-tracing?var-namespace=deepflow-otel-spring-demo&from=deepflow-doc) to see the results. -Summary of this tracing Demo: +Summary of this tracing demo: -- End-to-end: Integrated OTel, eBPF, and BPF, automatically traced 100 Spans of this Trace, including 20 eBPF Spans and 34 BPF Spans -- End-to-end: For services without OTel instrumentation, supports automatic tracing completion through eBPF, such as Spans 1-6 (loadgenerator) -- End-to-end: For services where OTel cannot be instrumented, supports automatic tracing completion through eBPF, such as eBPF Spans 67 and 100 depicting the start and end of MySQL Transactions (SET autocommit, commit) -- Full-stack: Supports tracing the network path between two Pods on the same K8s Node, such as Spans 91-92 -- Full-stack: Supports tracing the network path between two Pods across K8s Nodes, even if it passes through tunnel encapsulation, such as Spans 2-5 (IPIP tunnel encapsulation) -- Full-stack: eBPF and BPF Spans interspersed among OTel Spans, connecting applications, systems, and networks, such as eBPF Spans 12, 27, 41, 53 with their parent Spans (OTel) showing significant time differences that can be used to identify real performance bottlenecks, avoiding confusion among upstream and downstream application development teams +- End-to-end: Integrated OTel, eBPF, and BPF, automatically traced 100 Spans in this trace, including 20 eBPF Spans and 34 BPF Spans +- End-to-end: For services without OTel instrumentation, supports automatic tracing completion via eBPF, e.g., Spans 1-6 (loadgenerator) +- End-to-end: For services where OTel instrumentation is not possible, supports automatic tracing completion via eBPF, e.g., eBPF Spans 67 and 100 depict the start and end of a MySQL transaction (SET autocommit, commit) +- Full-stack: Supports tracing network paths between two Pods on the same K8s Node, e.g., Spans 91-92 +- Full-stack: Supports tracing network paths between Pods on different K8s Nodes, even when passing through tunnel encapsulation, e.g., Spans 2-5 (IPIP tunnel encapsulation) +- Full-stack: eBPF and BPF Spans interleave with OTel Spans, bridging application, system, and network. Significant time differences between eBPF Spans 12, 27, 41, 53 and their parent OTel Spans can help pinpoint real performance bottlenecks, avoiding confusion between upstream and downstream development teams # Experience with OpenTelemetry WebStore Demo -## Deploy Demo +## Deploy the Demo -This Demo is from [opentelemetry-webstore-demo](https://github.com/open-telemetry/opentelemetry-demo-webstore), which is a WebStore application composed of more than ten microservices implemented in languages such as Go, C#, Node.js, Python, and Java. Its application architecture is as follows: +This demo comes from [opentelemetry-webstore-demo](https://github.com/open-telemetry/opentelemetry-demo-webstore). +It consists of more than ten microservices implemented in Go, C#, Node.js, Python, Java, etc., with the following architecture: ```mermaid graph TD @@ -265,7 +272,7 @@ classDef erlang fill:#b83998,color:white; classDef php fill:#4f5d95,color:white; ``` -Deploy this Demo with one command: +Deploy the demo with one command: ```bash kubectl apply -n deepflow-otel-grpc-demo -f https://raw.githubusercontent.com/deepflowio/deepflow-demo/main/DeepFlow-Otel-Grpc-Demo/deepflow-otel-grpc-demo.yaml @@ -273,8 +280,10 @@ kubectl apply -n deepflow-otel-grpc-demo -f https://raw.githubusercontent.com/de ## View Tracing Data -Go to Grafana, open the `Distributed Tracing` Dashboard, select `namespace = deepflow-otel-grpc-demo`, and then choose a call to trace. DeepFlow can correlate tracing data obtained from OpenTelemetry, eBPF, and BPF in a single Trace flame graph, covering the full-stack call path of a multi-language application from business code, system functions, to network interfaces, achieving true end-to-end distributed tracing, as shown below: +Go to Grafana, open the `Distributed Tracing` Dashboard, select `namespace = deepflow-otel-grpc-demo`, and choose a request to trace. +DeepFlow can correlate tracing data from OpenTelemetry, eBPF, and BPF into a single trace flame graph, +covering the full-stack call path of a multi-language application from business code, system functions, to network interfaces, achieving true end-to-end distributed tracing, as shown below: ![OTel gRPC Demo](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/202208236304414496160.png) -You can also visit [DeepFlow Online Demo](https://ce-demo.deepflow.yunshan.net/d/Distributed_Tracing/distributed-tracing?var-namespace=deepflow-otel-grpc-demo&var-request_resource=*Order*&from=deepflow-doc) to see the effect. \ No newline at end of file +You can also visit the [DeepFlow Online Demo](https://ce-demo.deepflow.yunshan.net/d/Distributed_Tracing/distributed-tracing?var-namespace=deepflow-otel-grpc-demo&var-request_resource=*Order*&from=deepflow-doc) to see the results. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/02-tracing/05-skywalking.md b/translate/translated/08-integration/02-input/02-tracing/05-skywalking.md index 26bc3bea..b1b265ba 100644 --- a/translate/translated/08-integration/02-input/02-tracing/05-skywalking.md +++ b/translate/translated/08-integration/02-input/02-tracing/05-skywalking.md @@ -7,6 +7,8 @@ permalink: /integration/input/tracing/skywalking # Data Flow +## DeepFlow Community Edition + ```mermaid flowchart TD @@ -36,31 +38,64 @@ subgraph Host end ``` -# Configure OpenTelemetry SkyWalking Receiver +## DeepFlow Enterprise Edition -## Background Knowledge +```mermaid +flowchart TD -You can refer to the [OpenTelemetry documentation](https://opentelemetry.io/docs/) to understand the background knowledge of OpenTelemetry and refer to the previous section [OpenTelemetry Installation](../tracing/opentelemetry/#配置-opentelemetry) for quick installation of OpenTelemetry. +subgraph K8s-Cluster + subgraph AppPod + SWSDK1["sw-sdk / sw-javaagent"] + end + DeepFlowAgent1["deepflow-agent (daemonset)"] + DeepFlowServer["deepflow-server (deployment)"] -You can refer to the [SkyWalking documentation](https://skywalking.apache.org/docs/) to understand the background knowledge of SkyWalking. This demo does not require a full installation of SkyWalking; we will use OpenTelemetry to integrate SkyWalking's trace data. + SWSDK1 -->|sw-traces| DeepFlowAgent1 + DeepFlowAgent1 -->|sw-traces| DeepFlowServer +end -## Confirm OpenTelemetry Version +subgraph Host + subgraph AppProcess + SWSDK2["sw-sdk / sw-javaagent"] + end + DeepFlowAgent2[deepflow-agent] -First, you need to enable OpenTelemetry's ability to receive SkyWalking data, process the data through the OpenTelemetry standard protocol, and send it to the DeepFlow Agent. + SWSDK2 -->|sw-traces| DeepFlowAgent2 + DeepFlowAgent2 -->|sw-traces| DeepFlowServer +end +``` + +# Trace Collection + +## Collecting via DeepFlow Agent + +Starting from DeepFlow Enterprise Edition v6.6, DeepFlow Agent supports directly receiving and sending SkyWalking data without additional configuration. + +## Collecting via OpenTelemetry Collector + +### Background Knowledge + +You can refer to the [OpenTelemetry documentation](https://opentelemetry.io/docs/) to learn more about OpenTelemetry, and follow the [OpenTelemetry Installation](../tracing/opentelemetry/#配置-opentelemetry) section in the previous chapter to quickly install OpenTelemetry. + +You can refer to the [SkyWalking documentation](https://skywalking.apache.org/docs/) to learn more about SkyWalking. For this demo, you do not need to install the full SkyWalking stack — we will use OpenTelemetry to integrate SkyWalking trace data. -There is a bug in OpenTelemetry receiving SkyWalking data, which we have recently fixed in these two PRs [#11562](https://github.com/open-telemetry/opentelemetry-collector-contrib/pull/11562) and [#12651](https://github.com/open-telemetry/opentelemetry-collector-contrib/pull/12651). For the following demo, we need the OpenTelemetry [Collector image](https://hub.docker.com/r/otel/opentelemetry-collector-contrib) version `>= 0.57.0`. Please check the image version of the otel-agent in your environment and ensure it meets the requirements. Refer to the previous section [OpenTelemetry Installation](../tracing/opentelemetry/#配置-otel-agent) to update the otel-agent version in your environment. +### Confirm OpenTelemetry Version -## Configure OpenTelemetry to Receive SkyWalking Data +First, you need to enable OpenTelemetry’s ability to receive SkyWalking data, process it using the OpenTelemetry standard protocol, and then send it to the DeepFlow Agent. -After installing OpenTelemetry as described in the [Background Knowledge](#背景知识) section, we can configure OpenTelemetry to receive SkyWalking data using the following steps: +There is a bug in OpenTelemetry’s SkyWalking data receiver, which we have recently fixed in PRs [#11562](https://github.com/open-telemetry/opentelemetry-collector-contrib/pull/11562) and [#12651](https://github.com/open-telemetry/opentelemetry-collector-contrib/pull/12651). For the following demo, we require the OpenTelemetry [Collector image](https://hub.docker.com/r/otel/opentelemetry-collector-contrib) version `>= 0.57.0`. Please check the otel-agent image version in your environment and ensure it meets the requirement. You can refer to the [OpenTelemetry Installation](../tracing/opentelemetry/#配置-otel-agent) section in the previous chapter to update the otel-agent version in your environment. -Assuming the namespace where OpenTelemetry is located is `open-telemetry` and the ConfigMap used by otel-agent is named `otel-agent-conf`, use the following command to modify the otel-agent configuration: +### Configure OpenTelemetry to Receive SkyWalking Data + +As described in the [Background Knowledge](#背景知识) section, after installing OpenTelemetry, you can configure it to receive SkyWalking data using the following steps: + +Assume the namespace for OpenTelemetry is `open-telemetry`, and the ConfigMap used by otel-agent is named `otel-agent-conf`. Modify the otel-agent configuration with the following command: ```bash kubectl -n open-telemetry edit cm otel-agent-conf ``` -In the `receivers` section, add the following content: +In the `receivers` section, add the following: ```yaml receivers: @@ -73,7 +108,7 @@ receivers: endpoint: 0.0.0.0:12800 ``` -In the `service.pipelines.traces` section, add the following content: +In the `service.pipelines.traces` section, add the following: ```yaml service: @@ -83,37 +118,39 @@ service: receivers: [skywalking] ``` -At the same time, ensure that the `otel-agent-conf` has completed the corresponding configuration as described in the section [Configure otel-agent](../tracing/opentelemetry/#配置-otel-agent). +Also, make sure that `otel-agent-conf` has been configured according to the [Configure otel-agent](../tracing/opentelemetry/#配置-otel-agent) section. -Next, use the following command to modify the otel-agent Service to open the corresponding ports: +Next, modify the otel-agent Service to open the corresponding ports: ```bash kubectl -n open-telemetry patch service otel-agent -p '{"spec":{"ports":[{"name":"sw-http","port":12800,"protocol":"TCP","targetPort":12800},{"name":"sw-grpc","port":11800,"protocol":"TCP","targetPort":11800}]}}' ``` -Then, check the connection address configured in the application for the [SkyWalking OAP Server](https://skywalking.apache.org/docs/main/next/en/setup/backend/backend-setup/#requirements-and-default-settings) and modify it to the Service address of the Otel Agent: `otel-agent.open-telemetry`. For example, change the environment variable `SW_AGENT_COLLECTOR_BACKEND_SERVICES=oap-server:11800` to `SW_AGENT_COLLECTOR_BACKEND_SERVICES=otel-agent.open-telemetry:11800`. - -Of course, the reporting address configured in the application may take various forms. Please modify it according to the actual startup command of the application. For `Java` applications, just ensure that the address injected in the startup command can be modified, such as: `-Dskywalking.collector.backend_service=otel-agent.open-telemetry:11800`. - -Finally, restart the otel-agent to complete the update: +Finally, restart the otel-agent to apply the update: ```bash kubectl rollout restart -n open-telemetry daemonset/otel-agent ``` +# Modify SkyWalking Sending Configuration + +Finally, check the configured [SkyWalking OAP Server](https://skywalking.apache.org/docs/main/next/en/setup/backend/backend-setup/#requirements-and-default-settings) address in your application, and change it to the Otel Agent Service address: `otel-agent.open-telemetry`. For example, change the environment variable `SW_AGENT_COLLECTOR_BACKEND_SERVICES=oap-server:11800` to `SW_AGENT_COLLECTOR_BACKEND_SERVICES=otel-agent.open-telemetry:11800`. If you are using DeepFlow Agent to receive data directly, change it to `deepflow-agent.deepflow`. + +Of course, the reporting address in the application configuration may take various forms. Please modify it according to the actual application startup command. For `Java` applications, you only need to ensure that the injected address in the startup command is modified, for example: `-Dskywalking.collector.backend_service=otel-agent.open-telemetry:11800`. + # Configure DeepFlow -Please refer to the section [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) to complete the configuration of the DeepFlow Agent. +Please refer to the [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) section to complete the DeepFlow Agent configuration. -# Experience Based on WebShop Demo +# Experience with WebShop Demo -## Deploy Demo +## Deploy the Demo -This demo comes from [this GitHub repository](https://github.com/liuzhibin-cn/my-demo). It is a WebShop application composed of five microservices written in Spring Boot. Its architecture is as follows: +This demo comes from [this GitHub repository](https://github.com/liuzhibin-cn/my-demo). It is a WebShop application built with Spring Boot, consisting of five microservices. Its architecture is as follows: ![Sping Boot Demo Architecture](./imgs/spring-boot-webshop-arch.png) -You can deploy this demo with one click using the following command. This demo has already configured the reporting address, so no additional modifications are needed. +You can deploy this demo with a single command. The reporting address has already been configured, so no further modification is needed. ```bash kubectl apply -f https://raw.githubusercontent.com/deepflowio/deepflow-demo/main/DeepFlow-Otel-SkyWalking-Demo/deepflow-otel-skywalking-demo.yaml @@ -121,10 +158,10 @@ kubectl apply -f https://raw.githubusercontent.com/deepflowio/deepflow-demo/main ## View Tracing Data -Go to Grafana, open the `Distributed Tracing` Dashboard, select `namespace = deepflow-otel-skywalking-demo`, and then you can choose a call to trace. -DeepFlow can correlate and display the tracing data obtained from SkyWalking, eBPF, and BPF in a single trace flame graph, -covering the full-stack call path of a Spring Boot application from business code, system functions, to network interfaces, achieving true full-link distributed tracing, as shown below: +Go to Grafana, open the `Distributed Tracing` dashboard, select `namespace = deepflow-otel-skywalking-demo`, and then choose a call to trace. +DeepFlow can correlate and display tracing data from SkyWalking, eBPF, and BPF in a single trace flame graph, +covering the full-stack call path of a Spring Boot application from business code, system functions, to network interfaces, achieving true end-to-end distributed tracing. The result looks like this: ![OTel SkyWalking Demo](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/2022082363044145adc1b.png) -You can also visit the [DeepFlow Online Demo](https://ce-demo.deepflow.yunshan.net/d/Distributed_Tracing/distributed-tracing?var-namespace=deepflow-otel-skywalking-demo&from=deepflow-doc) to see the effect. +You can also visit the [DeepFlow Online Demo](https://ce-demo.deepflow.yunshan.net/d/Distributed_Tracing/distributed-tracing?var-namespace=deepflow-otel-skywalking-demo&from=deepflow-doc) to see the effect. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/03-profile/01-profile-auto-tagging.md b/translate/translated/08-integration/02-input/03-profile/01-profile-auto-tagging.md index eebf0612..0146243a 100644 --- a/translate/translated/08-integration/02-input/03-profile/01-profile-auto-tagging.md +++ b/translate/translated/08-integration/02-input/03-profile/01-profile-auto-tagging.md @@ -5,6 +5,6 @@ permalink: /integration/input/profile/profile-auto-tagging > This document was translated by ChatGPT -DeepFlow Agent needs to inject IP tags into the data so that DeepFlow Server can expand all other tags based on this. For continuous profiling data, when the data is sent to the DeepFlow Agent via the HTTP protocol, it will obtain the upstream client IP and inject it into the data. Reducing data transfer can improve matching accuracy, for example, by sending it directly to the service without going through an intermediate proxy when sending to the DeepFlow Agent. +The DeepFlow Agent needs to inject IP tags into the data so that the DeepFlow Server can derive all other tags based on them. For continuous profiling data, when the data is sent to the DeepFlow Agent via the HTTP protocol, the upstream client IP will be obtained and injected into the data. Reducing data relays can improve matching accuracy — for example, sending directly to the DeepFlow Agent instead of going through an intermediate proxy before reaching the service. -Based on DeepFlow's AutoTagging capability, we have automated the association of tracing data with other observability data, without requiring developers to insert any code. +With DeepFlow’s AutoTagging capability, we can automatically associate tracing data with other observability data, without requiring developers to insert any code. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/03-profile/02-profile.md b/translate/translated/08-integration/02-input/03-profile/02-profile.md index 0ac4d5e1..7331e019 100644 --- a/translate/translated/08-integration/02-input/03-profile/02-profile.md +++ b/translate/translated/08-integration/02-input/03-profile/02-profile.md @@ -32,15 +32,15 @@ subgraph Host end ``` -# Configure Application +# Configure the Application -Currently, DeepFlow supports Profile data (continuous profiling data) integration based on [Pyroscope](https://github.com/grafana/pyroscope) and Golang pprof, Java Jfr. +Currently, DeepFlow supports integration of Profile data (continuous profiling data) based on [Pyroscope](https://github.com/grafana/pyroscope), Golang pprof, and Java Jfr. ## Based on Pyroscope SDK -DeepFlow currently supports Profile data sent via Pyroscope SDK. You can find the supported language SDKs in the [Pyroscope SDKs](https://grafana.com/docs/pyroscope/latest/configure-client/#pyroscope-sdks-sdk-instrumentation) documentation and complete the instrumentation in your code. +DeepFlow currently supports receiving Profile data sent via the Pyroscope SDK. You can find SDKs for supported languages in the [Pyroscope SDKs](https://grafana.com/docs/pyroscope/latest/configure-client/#pyroscope-sdks-sdk-instrumentation) documentation and instrument your code accordingly. -After modifying the code, change the target sending address to DeepFlow Agent via environment variables in the application's runtime environment. For example, in a K8S deployment, add the following environment variables to the deployment file: +After modifying the code, set the target sending address to the DeepFlow Agent via environment variables in the application runtime environment. For example, in a K8S deployment, add the following environment variables to the deployment file: ```yaml env: @@ -48,7 +48,7 @@ env: value: http://deepflow-agent.deepflow/api/v1/profile ``` -Additionally, to identify different data sources in DeepFlow, explicitly label the application name by adding the following environment variables: +In addition, to identify different data sources in DeepFlow, you need to explicitly specify the application name by adding the following environment variable: ```yaml env: @@ -58,22 +58,22 @@ env: ## Based on Golang pprof -Profile data collected based on Golang pprof can also be sent to DeepFlow, but you need to manually add the code logic for active reporting. You can collect Profile data via ["net/http/pprof"](https://pkg.go.dev/net/http/pprof) and expose the data download service. After the client requests the `/debug/pprof/profile` interface to get the target pprof data, send it to DeepFlow. +Profile data collected via Golang pprof can also be sent to DeepFlow, but you need to manually add the logic for active reporting. You can use ["net/http/pprof"](https://pkg.go.dev/net/http/pprof) to collect Profile data and expose a data download service. After the client requests the `/debug/pprof/profile` endpoint to obtain the target pprof data, send it to DeepFlow. -Here is a reference code for building the reporting logic: +Below is a sample code snippet for building the reporting logic: ```go func main() { - // Note, the actual URL used here is `/api/v1/profile/ingest` + // Note: the actual URL used here is `/api/v1/profile/ingest` deepflowAgentAddress = "http://deepflow-agent.deepflow/api/v1/profile/ingest" - var pprof io.Reader // FIXME: This is an example, please first get the pprof data from `/debug/pprof/profile` + var pprof io.Reader // FIXME: This is an example, please first obtain pprof data from `/debug/pprof/profile` err = sendProfileData(pprof, deepflowAgentAddress) if err != nil { fmt.Println(err) } } -// Build the request to send +// Build and send the request func sendProfileData(pprof io.Reader, remoteURL string) error { bodyBuf := &bytes.Buffer{} bodyWriter := multipart.NewWriter(bodyBuf) @@ -100,10 +100,10 @@ func sendProfileData(pprof io.Reader, remoteURL string) error { q := u.Query() q.Set("spyName", "gospy") // hardcode, no need to modify q.Set("name", "application-demo") // FIXME: your application name - q.Set("unit", "samples"); // FIXME: unit of measurement, see explanation below + q.Set("unit", "samples"); // FIXME: unit, see explanation below q.Set("from", strconv.Itoa(int(time.Now().Unix()))) // FIXME: profile start time q.Set("until", strconv.Itoa(int(time.Now().Unix()))) // FIXME: profile end time - q.Set("sampleRate", "100") // FIXME: actual sampling rate of the profile, sampling rate 100=1s/10ms, i.e., sampled every 10ms + q.Set("sampleRate", "100") // FIXME: actual profile sampling rate, 100=1s/10ms, i.e., sample every 10ms u.RawQuery = q.Encode() req, err := http.NewRequest(http.MethodPost, u.String(), bodyBuf) @@ -130,9 +130,9 @@ func sendProfileData(pprof io.Reader, remoteURL string) error { ## Based on Java Async Profiler -For Java applications, we support receiving Profile data in [JFR](https://docs.oracle.com/javacomponents/jmc-5-4/jfr-runtime-guide/about.htm) format. Use Java's built-in [jcmd](https://docs.oracle.com/javase/8/docs/technotes/guides/troubleshoot/tooldescr006.html) or [async-profiler](https://github.com/async-profiler/async-profiler) to collect Profile data and generate Jfr format to send to DeepFlow. +For Java applications, we support receiving Profile data in [JFR](https://docs.oracle.com/javacomponents/jmc-5-4/jfr-runtime-guide/about.htm) format. Use Java's built-in [jcmd](https://docs.oracle.com/javase/8/docs/technotes/guides/troubleshoot/tooldescr006.html) or [async-profiler](https://github.com/async-profiler/async-profiler) to collect Profile data, generate Jfr format, and send it to DeepFlow. -Here is a reference code for building the reporting logic: +Below is a sample code snippet for building the reporting logic: ```java import okhttp3.*; @@ -143,7 +143,7 @@ import okio.Okio; import java.io.IOException; public class Sender { - // Note, the actual URL used here is `/api/v1/profile/ingest` + // Note: the actual URL used here is `/api/v1/profile/ingest` private static final String DEEPFLOW_AGENT_ADDRESS = "http://deepflow-agent.deepflow/api/v1/profile/ingest"; public static void main(String[] args) throws IOException { @@ -153,7 +153,7 @@ public class Sender { private static void sendProfileData(String remoteURL) throws IOException { OkHttpClient client = new OkHttpClient(); - byte[] data = new byte[] {}; // FIXME: This is an example, please use the profile tool to get the Jfr content before reporting + byte[] data = new byte[] {}; // FIXME: This is an example, please use a profiling tool to obtain Jfr content before reporting MediaType mediaType = MediaType.parse("application/octet-stream"); RequestBody requestBody = new RequestBody() { @@ -162,7 +162,7 @@ public class Sender { return mediaType; } - // Use Gzip for compression before transmission + // Use Gzip compression for transmission @Override public void writeTo(BufferedSink sink) throws IOException { BufferedSink gzipSink = Okio.buffer(new GzipSink(sink)); @@ -175,10 +175,10 @@ public class Sender { urlBuilder.addQueryParameter("name", "application-demo"); // FIXME: your application name urlBuilder.addQueryParameter("spyName", "javaspy"); // hardcode, no need to modify urlBuilder.addQueryParameter("format", "jfr"); // hardcode, no need to modify - urlBuilder.addQueryParameter("unit", "samples"); // FIXME: unit of measurement, see explanation below + urlBuilder.addQueryParameter("unit", "samples"); // FIXME: unit, see explanation below urlBuilder.addQueryParameter("from", String.valueOf(System.currentTimeMillis() / 1000)); // FIXME: profile start time urlBuilder.addQueryParameter("until", String.valueOf(System.currentTimeMillis() / 1000));// FIXME: profile end time - urlBuilder.addQueryParameter("sampleRate", "100"); // FIXME: actual sampling rate of the profile, sampling rate 100=1s/10ms, i.e., sampled every 10ms + urlBuilder.addQueryParameter("sampleRate", "100"); // FIXME: actual profile sampling rate, 100=1s/10ms, i.e., sample every 10ms String urlWithQueryParams = urlBuilder.build().toString(); Request request = new Request.Builder() @@ -197,25 +197,25 @@ public class Sender { } ``` -## Reporting Parameter Explanation +## Reporting Parameter Description | Name | Type | Description | -| ---------- | ------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| name | string | Application name, used to identify the reported data. You can add custom tags to mark different deployment specifications of the same application, e.g., `application-demo{region="cn",deploy="prod"}` | -| spyName | string | Used to mark the type of reported data. For Golang applications, it is fixed as `gospy`, and for Java applications, it is fixed as `javaspy` | -| format | string | Profile data format. For Golang, the collected pprof format is `pprof` (default), and for Java, the collected jfr format is `jfr` | -| unit | string | Unit of measurement. For different sampling types, there are different units. Refer to [here](https://github.com/deepflowio/deepflow/blob/v6.4.9/server/ingester/profile/dbwriter/profile.go#L99) for specific meanings: `cpu` uses `samples` as the unit, `memory` uses `bytes` as the unit, and others are similar | +| ---------- | ------ | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| name | string | Application name, used to identify reported data. You can also add custom labels to distinguish different deployment specs of the same application, e.g., `application-demo{region="cn",deploy="prod"}` | +| spyName | string | Used to indicate the type of reported data. For Golang applications, fixed as `gospy`; for Java applications, fixed as `javaspy` | +| format | string | Profile data format. Golang pprof format is `pprof` (default), Java Jfr format is `jfr` | +| unit | string | Unit of measurement. Different sampling types have different units. See [here](https://github.com/deepflowio/deepflow/blob/v6.4.9/server/ingester/profile/dbwriter/profile.go#L99) for details. For `cpu`, use `samples`; for `memory`, use `bytes`; others are similar | | from | int | Profile start time, Unix timestamp (seconds) | | until | int | Profile end time, Unix timestamp (seconds) | -| sampleRate | int | Actual sampling rate of the profile | +| sampleRate | int | Actual profile sampling rate | # Configure DeepFlow -Refer to the [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) section to complete the configuration of DeepFlow Agent and open the data integration port. +Refer to the [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) section to complete the DeepFlow Agent configuration and enable the data integration port. -# Experience Based on Demo +# Experience with Demo -Use the following command to quickly deploy the demo and experience the continuous profiling capability in DeepFlow: +Use the following commands to quickly deploy a demo and experience the continuous profiling capability in DeepFlow: ::: code-tabs#shell @@ -233,4 +233,4 @@ kubectl apply -f https://raw.githubusercontent.com/deepflowio/deepflow-demo/main ::: -Then, for the community edition, refer to the [Continuous Profiling - View Data](../../../features/continuous-profiling/data/) section to obtain the data generated by continuous profiling. \ No newline at end of file +Then, for the community edition, refer to the [Continuous Profiling - View Data](../../../features/continuous-profiling/data/) section to access the data generated by continuous profiling. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/04-log/01-log-auto-tagging.md b/translate/translated/08-integration/02-input/04-log/01-log-auto-tagging.md index f65dbb41..f2a15edd 100644 --- a/translate/translated/08-integration/02-input/04-log/01-log-auto-tagging.md +++ b/translate/translated/08-integration/02-input/04-log/01-log-auto-tagging.md @@ -5,6 +5,6 @@ permalink: /integration/input/log/log-auto-tagging > This document was translated by ChatGPT -DeepFlow Agent needs to inject IP tags into the data so that DeepFlow Server can expand all other tags based on this. For log data using the [Kubernetes_Log](https://vector.dev/docs/reference/configuration/sources/kubernetes_logs/) module of Vector, it automatically discovers and attaches the PodName of the log source in the log stream, allowing DeepFlow to automatically match the data source. +The DeepFlow Agent needs to inject IP tags into the data so that the DeepFlow Server can use them to derive all other tags. For log data using the [Kubernetes_Log](https://vector.dev/docs/reference/configuration/sources/kubernetes_logs/) module of Vector, it automatically discovers and appends the PodName of the log source to the log stream, enabling DeepFlow to automatically match the data source. -Based on DeepFlow's AutoTagging capability, we have automated the association of tracing data with other observability data, without requiring developers to insert any code. +With DeepFlow’s AutoTagging capability, we automate the association between tracing data and other observability data, without requiring developers to insert any code. \ No newline at end of file diff --git a/translate/translated/08-integration/02-input/04-log/02-vector.md b/translate/translated/08-integration/02-input/04-log/02-vector.md index c8fa0d64..d10d2d78 100644 --- a/translate/translated/08-integration/02-input/04-log/02-vector.md +++ b/translate/translated/08-integration/02-input/04-log/02-vector.md @@ -33,8 +33,8 @@ end ## Install Vector -You can learn the relevant background knowledge in the [Vector documentation](https://vector.dev/docs/). -If your cluster does not have Vector, you can deploy Vector using the following steps: +You can learn the relevant background knowledge in the [Vector documentation](https://vector.dev/docs/). +If your cluster does not have Vector, you can deploy it using the following steps: ::: code-tabs#shell @@ -133,7 +133,7 @@ customConfig: if !exists(.level) { if exists(.json) { - .level = .json.level + .level = to_string!(.json.level) del(.json.level) } else { # match log levels surround by ``[]`` or ``<>`` with ignore case @@ -174,7 +174,7 @@ helm install vector vector/vector \ ::: -Before configuring, you can first understand the [Vector workflow](https://vector.dev/docs/about/under-the-hood/architecture/pipeline-model/), where data flows through the following modules in sequence, from the source to the destination: +Before configuration, you can first learn about the [Vector workflow](https://vector.dev/docs/about/under-the-hood/architecture/pipeline-model/). Data flows in the following module order, from the source to the destination: ```mermaid flowchart LR @@ -187,9 +187,66 @@ Source -->|log| Transform Transform -->|log| Sink ``` +Generally, a Vector configuration contains at least the `sources` module and the `sinks` module. If additional data processing is required, you must add the `transforms` module to clean the data into the final desired content. A typical Vector configuration looks like this: + +```yaml +# Data sources +sources: + nginx_logs: + # ... + file_logs: + # ... + kubernetes_logs: + # ... + +# Data processing +transforms: + tag_log: + # inputs refers to the data source; here you can configure the key from sources or from other transforms + inputs: + - nginx_logs + # ... + flush_log: + # tag_log comes from the previous transforms module, so the same data is processed sequentially by two transforms modules + inputs: + - tag_log + - file_logs + # ... + +# Data output +sinks: + push_log: + # Similarly, inputs here can come from both sources and transforms modules + inputs: + - flush_log + - kubernetes_logs + # ... +``` + +In the above example, different configurations implement three different data flows: + +```mermaid +flowchart LR + +NginxLog["nginx_logs"] +FileLog["file_logs"] +KubernetesLog["kubernetes_logs"] +TagLog["tag_log"] +FlushLog["flush_log"] +PushLog["push_log"] + +NginxLog --> TagLog +FileLog --> FlushLog +KubernetesLog --> PushLog +TagLog --> FlushLog +FlushLog --> PushLog +``` + +Next, let's look at the specific configuration of each module. + ## Collect Logs -After installing Vector, we can use the [Kubernetes_Log](https://vector.dev/docs/reference/configuration/sources/kubernetes_logs/) module to collect logs from Pods deployed in Kubernetes. Since DeepFlow has already actively learned the labels and annotations related to Pods in Kubernetes through the AutoTagging mechanism, the log stream can be sent without this part to reduce transmission volume. The sample configuration is as follows: +After installing Vector, we can use the [Kubernetes_Log](https://vector.dev/docs/reference/configuration/sources/kubernetes_logs/) module to obtain logs from Pods deployed in Kubernetes. Since DeepFlow has already learned the relevant Labels and Annotations of Pods in Kubernetes through the AutoTagging mechanism, you can remove this part of the content when sending log streams to reduce transmission volume. Example configuration: ```yaml sources: @@ -204,7 +261,7 @@ sources: pod_labels: '' ``` -If you deploy Vector as a process on a cloud server, you can use the [File](https://vector.dev/docs/reference/configuration/sources/file) module to collect logs from a specified path. Taking the `/var/log/` path as an example, the sample configuration is as follows: +If you deploy Vector as a process on a cloud server, you can use the [File](https://vector.dev/docs/reference/configuration/sources/file) module to obtain logs from a specified path. Using `/var/log/` as an example: ```yaml sources: @@ -214,7 +271,7 @@ sources: - /var/log/*.log - /var/log/**/*.log exclude: - # FIXME: If both kubernetes_logs and file modules are configured, to avoid duplicate log content, remove the k8s log folder + # FIXME: If both kubernetes_logs and file modules are configured, remove k8s log folders to avoid duplicate monitoring - /var/log/pods/** - /var/log/containers/** fingerprint: @@ -223,7 +280,7 @@ sources: ## Inject Tags -Then, we can use the [Remap](https://vector.dev/docs/reference/configuration/transforms/remap/) module in Transforms to add necessary tags to the sent logs. Currently, we require these two tags: `_df_log_type` and `level`. Below is a sample configuration: +We can then use the [Remap](https://vector.dev/docs/reference/configuration/transforms/remap/) module in Transforms to add necessary tags to the logs being sent. Currently, we require two tags: `_df_log_type` and `level`. Example configuration: ```yaml transforms: @@ -243,7 +300,7 @@ transforms: if !exists(.level) { if exists(.json) { - .level = .json.level + .level = to_string!(.json.level) del(.json.level) } else { # match log levels surround by `[]` or `<>` with ignore case @@ -265,22 +322,22 @@ transforms: } if !exists(.app_service) { - # FIXME: files module does not have this field, please inject the application name through the log content + # FIXME: files module does not have this field, please inject application name via log content .app_service = .kubernetes.container_name } ``` -In this code snippet, we assume that we may get both JSON formatted log content and non-JSON formatted log content. For both types of logs, we try to extract their log level `level`. For JSON formatted logs, we extract their content to the outer `message` field and put all remaining JSON keys into a field named `json`. At the end of this code, we tag both types of logs with `_df_log_type=user` and `app_service=kubernetes.container_name`. +In this snippet, we assume that we may get both JSON-formatted logs and non-JSON logs. For both types, we try to extract the log level `level`. For JSON logs, we extract its content into the outer `message` field and put the remaining JSON keys into a field named `json`. At the end, we add `_df_log_type=user` and `app_service=kubernetes.container_name` tags to both types of logs. -If there are richer log formats that need to be matched in actual use, you can refer to the [Vrl](https://vector.dev/docs/reference/vrl/) syntax rules to customize your log extraction rules. +If you have richer log formats to match in practice, refer to the [Vrl](https://vector.dev/docs/reference/vrl/) syntax rules to customize your log extraction rules. ## Common Configurations -In addition to the above configurations, the Transforms module can also implement many features to help us get more accurate information from the logs. Here are some common configurations: +In addition to the above, the Transforms module can implement many features to help extract more accurate information from logs. Here are some common configurations: ### Merge Multi-line Logs -Usage suggestion: Use regex to match the "start pattern" of the log. Before encountering the next "start pattern", all logs are aggregated into one log message and retain the newline character. To reduce mismatches, use a date-time format like `yyyy-MM-dd HH:mm:ss` to match the beginning of a log line. +Recommendation: Use regex to match the "start pattern" of a log. Before encountering the next "start pattern", aggregate all logs into one message and keep line breaks. To reduce mismatches, match a datetime format like `yyyy-MM-dd HH:mm:ss` at the start of a log line. ```yaml transforms: @@ -301,7 +358,7 @@ transforms: ### Filter Color Control Characters -Usage suggestion: Use regex to filter color control characters in the log to increase log readability. +Recommendation: Use regex to filter color control characters in logs to improve readability. ```yaml transforms: @@ -314,9 +371,9 @@ transforms: .message = replace(string!(.message), r'\u001B\[([0-9]{1,3}(;[0-9]{1,3})*)?m', "") ``` -### Extract Log Levels +### Extract Log Level -Usage suggestion: Use regex to try to match the log levels that appear in the log. To reduce mismatches, symbols like `[]` can be added around the log level. +Recommendation: Use regex to try to match log levels in logs. To reduce mismatches, you can enclose log levels in symbols like `[]`. ```yaml transforms: @@ -340,7 +397,7 @@ transforms: ### Extract Custom Tags -If the application needs to inject some custom tags for filtering logs, similarly, you can use the Remap module of Transforms to write a piece of code to inject tags. We require custom tags to be written into the `.json` structure to be stored and queried. The example is as follows: +If your application needs to inject some custom tags for log filtering, you can also use the Remap module in Transforms to write code to inject tags. We require that custom tags must be written into the `.json` struct to be stored and queried. Example: ```yaml transforms: @@ -352,11 +409,11 @@ transforms: source: |- .json = { "cluster": "Production", - "extra_user_tag": "xxxxx" # FIXME: Customize the tags you need to add + "extra_user_tag": "xxxxx" # FIXME: customize the tags you need } ``` -Then, when using the [SQL API](../../output/query/sql) for querying, you can use the following statement to filter the injected tags: +Then, when using the [SQL API](../../output/query/sql) to query, you can filter the injected tags with: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ @@ -379,11 +436,11 @@ sinks: uri: http://deepflow-agent.deepflow/api/v1/log ``` -Combining these three modules together, you can collect logs, inject tags, and finally send them to DeepFlow. +By combining these three modules, you can collect logs, inject tags, and finally send them to DeepFlow. ## Complete Example -Based on the above explanation, we provide a complete example. Assuming the collection target is an **nginx application deployed on a cloud server**, you can collect its logs and send them to DeepFlow with the following configuration: +Based on the above, here is a complete example. Suppose the collection target is an **nginx application deployed on a cloud server**, you can collect its logs and send them to DeepFlow with the following configuration: ```yaml sources: @@ -413,10 +470,10 @@ sinks: codec: json inputs: - tag_nginx_log - uri: http://${deepflow-agent-host}:${port}/api/v1/log # FIXME: Fill in the target DeepFlow Agent address that can receive data here + uri: http://${deepflow-agent-host}:${port}/api/v1/log # FIXME: Fill in the address of the target DeepFlow Agent that can receive data type: http ``` # Configure DeepFlow -To allow the DeepFlow Agent to receive this part of the data, please refer to the [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) section to complete the DeepFlow Agent configuration. \ No newline at end of file +To allow the DeepFlow Agent to receive this data, please refer to the [Configure DeepFlow](../tracing/opentelemetry/#配置-deepflow) section to complete the DeepFlow Agent configuration. \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/01-query/01-sql.md b/translate/translated/08-integration/03-output/01-query/01-sql.md index 121023ef..4d378eb1 100644 --- a/translate/translated/08-integration/03-output/01-query/01-sql.md +++ b/translate/translated/08-integration/03-output/01-query/01-sql.md @@ -7,9 +7,9 @@ permalink: /integration/output/query/sql # Introduction -Provides a unified SQL interface to query all types of observability data. It can be used as a DataSource for Grafana or to implement your own GUI based on it. +Provides a unified SQL interface to query all types of observability data. It can be used as a Grafana DataSource or as the basis for building your own GUI. -# SQL Server Endpoint +# SQL Service Endpoint Get the service endpoint port number: @@ -27,14 +27,14 @@ SQL statement: show databases ``` -API call method: +API call example: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ --data-urlencode "sql=show databases" ``` -## Get All Tables in a Specified Database +## Get All Tables in a Specific Database SQL statement: @@ -42,7 +42,7 @@ SQL statement: show tables ``` -API call method: +API call example: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ @@ -50,7 +50,7 @@ curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ --data-urlencode "sql=show tables" ``` -## Get Tags in a Specified Table +## Get Tags in a Specific Table SQL statement: @@ -58,7 +58,7 @@ SQL statement: show tags from ${table_name} ``` -API call method: +API call example: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ @@ -66,7 +66,7 @@ curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ --data-urlencode "sql=show tags from ${table_name}" ``` -Output example: +Example output: ```json { @@ -81,12 +81,12 @@ Output example: "type" // int, int_enum, string, string_enum, resource_name, resource_id, ip ], "values": [ - ["chost", "chost_0", "chost_1", "云服务器", "resource_id"], + ["chost", "chost_0", "chost_1", "Cloud Server", "resource_id"], [ "chost_name", "chost_name_0", "chost_name_1", - "云服务器名称", + "Cloud Server Name", "resource_name" ] ] @@ -94,7 +94,7 @@ Output example: } ``` -## Get Values of a Specified Tag +## Get Values of a Specific Tag ### Get All Values of a Tag @@ -104,13 +104,13 @@ SQL statement: show tag ${tag_name} values from ${table_name} ``` -The above statement can also use `limit` and `offset` keywords to reduce the number of returned values: +The above statement can also use the `limit` and `offset` keywords to reduce the number of returned values: ```SQL show tag ${tag_name} values from ${table_name} limit 100 offset 100 ``` -API call method: +API call example: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ @@ -118,7 +118,7 @@ curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ --data-urlencode "sql=show tag ${tag_name} values from ${table_name}" ``` -Output example: +Example output: ```json { @@ -131,9 +131,9 @@ Output example: } ``` -### Filter Using Tag's Own Name +### Filter Using the Tag’s Own Fields -Note that the values of the above Tag will return three columns: `value`, `display_name`, `uid`. We can use this information for filtering, for example: +Note that the returned values of a Tag contain three columns: `value`, `display_name`, and `uid`. We can use this information for filtering, for example: ```SQL show tag ${tag_name} values from ${table_name} where display_name like '*abc*' @@ -141,29 +141,35 @@ show tag ${tag_name} values from ${table_name} where display_name like '*abc*' API call method and output example are the same as above. -### Filter Using Other Tags +### Filter by Associating with Other Tags -Sometimes we want to use tags for associative filtering to reduce the range of candidate values. In this case, we can choose to query a data table, filter by Tag1, and aggregate by Tag2. For example, we want to query all `pod` names in `pod_cluster="cluster1"`: +Sometimes we want to filter candidate values by associating with another Tag. In this case, we can query a table, filter by Tag1, and aggregate by Tag2. +For example, to query all `pod` names in `pod_cluster="cluster1"`: ```SQL SELECT pod FROM `network.1m` WHERE pod_cluster = 'cluster1' GROUP BY pod ``` -The above statement will use the `pod_cluster` field in the `network.1m` table of the `flow_metrics` database to filter and group the candidate `pod` values. Of course, we can also achieve this by querying any table in DeepFlow, but we should avoid using tables with large amounts of data. Additionally, we can add other dimensions such as time in the SQL to speed up the search: +The above statement uses the `pod_cluster` field in the `network.1m` table of the `flow_metrics` database to filter and group `pod` candidates. +Of course, we can query any table in DeepFlow to achieve this, but it’s best to avoid tables with very large data volumes. +We can also add time or other dimensions in SQL to speed up the search: ```SQL SELECT pod FROM `network.1m` WHERE pod_cluster = 'cluster1' AND time > 1234567890 GROUP BY pod ``` -Note: Data can only be found in `flow_metrics` if the pod has had traffic (and is not using HostNetwork). After integrating Prometheus or Telegraf data, we can also use the constant metrics in them to assist in obtaining Tag values. For example, we can use Prometheus metrics in `ext_metrics` to achieve the above requirement: +Note: You can only find data in `flow_metrics` if the pod has had traffic before (and is not using HostNetwork). +After integrating Prometheus or Telegraf data, we can also use constant metrics to help retrieve Tag values. +For example, we can use Prometheus metrics in `ext_metrics` to achieve the above: ```SQL SELECT pod FROM `prometheus.kube_pod_start_time` WHERE pod_cluster = 'cluster1' GROUP BY pod ``` -In Grafana, we can also use the above capabilities to achieve linked filtering of Variable candidates. For example, we use a custom Variable $cluster and built-in Variables [`$__from`, `$__to`](https://grafana.com/docs/grafana/latest/dashboards/variables/add-template-variables/#__from-and-__to) to perform linked filtering on another Variable pod: +In Grafana, we can also use this capability to implement linked filtering for Variable candidates. +For example, using a custom Variable `$cluster` and built-in Variables [`$__from`, `$__to`](https://grafana.com/docs/grafana/latest/dashboards/variables/add-template-variables/#__from-and-__to) to filter another Variable `pod`: -- When the value of cluster is id, use `$cluster`: +- When the cluster value is an id, use `$cluster`: ```Bash cluster = [1, 2] @@ -171,11 +177,11 @@ In Grafana, we can also use the above capabilities to achieve linked filtering o ``` ```SQL - // Add 5 minutes before and after the time range to avoid frequent changes of candidates + -- Add 5 minutes before and after the time range to avoid frequent changes of candidates SELECT pod_id as `value`, pod as `display_name` FROM `network.1m` WHERE pod_cluster IN ($cluster) AND time >= ${__from:date:seconds}-500 AND time <= ${__to:date:seconds}+500 GROUP BY `value` ``` -- When the value of cluster is name, use `${cluster:singlequote}`: +- When the cluster value is a name, use `${cluster:singlequote}`: ```Bash cluster = [deepflow-a, deepflow-b] @@ -186,7 +192,7 @@ In Grafana, we can also use the above capabilities to achieve linked filtering o SELECT pod as `value`, pod as `display_name` FROM `network.1m` WHERE pod_cluster IN (${cluster:singlequote}) AND time >= ${__from:date:seconds}-500 AND time <= ${__to:date:seconds}+500 GROUP BY `value` ``` -## Get Metrics in a Specified Table +## Get Metrics in a Specific Table SQL statement: @@ -194,7 +200,7 @@ SQL statement: show metrics from ${table_name} ``` -API call method: +API call example: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ @@ -216,7 +222,7 @@ ORDER BY col_3 \ LIMIT 100 ``` -API call method: +API call example: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ @@ -224,26 +230,27 @@ curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ --data-urlencode "sql=${sql}" ``` -When `db=flow_metric`, you need to specify the data precision through `--data-urlencode "data_precision=${data_precision}"`. The optional values for `data_precision` are `1m` and `1s`. +When `db=flow_metric`, you need to specify the data precision with `--data-urlencode "data_precision=${data_precision}"`. +The optional values for `data_precision` are `1m` and `1s`. # SQL Query Functions -## Functions Supported by Tags +## Functions Supported for Tags -- enum - - Description: `enum(observation_point)` converts enum fields to values - - Example: `SELECT enum(observation_point) ...`, `... WHERE enum(observation_point) = 'xxx' ...` - - Note: Only `string_enum` and `int_enum` types of Tags are supported +- enum + - Description: `enum(observation_point)` converts an enum field to its value + - Example: `SELECT enum(observation_point) ...`, `... WHERE enum(observation_point) = 'xxx' ...` + - Note: Only `string_enum` and `int_enum` type Tags are supported -## Functions Supported by Metrics +## Functions Supported for Metrics -Execute the following SQL statement to get all functions: +Run the following SQL statement to get all functions: ```SQL show metric function ``` -API call method: +API call example: ```bash curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ @@ -252,5 +259,5 @@ curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/query/" \ # SQL Syntax -- Left values do not support spaces, single quotes, or backticks -- Single quotes in right values need to be escaped with the escape character `\` +- The left-hand side does not support spaces, single quotes, or backticks +- Single quotes in the right-hand side need to be escaped with `\` \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/01-query/02-promql.md b/translate/translated/08-integration/03-output/01-query/02-promql.md index 4a65194d..427e78df 100644 --- a/translate/translated/08-integration/03-output/01-query/02-promql.md +++ b/translate/translated/08-integration/03-output/01-query/02-promql.md @@ -7,18 +7,18 @@ permalink: /integration/output/query/promql # Introduction -Starting from v6.2.1, DeepFlow supports PromQL. The following Prometheus APIs are currently implemented and can be called directly via HTTP as per the [Prometheus API definition](https://prometheus.io/docs/prometheus/latest/querying/api/#expression-queries): +DeepFlow has supported PromQL since v6.2.1. The following Prometheus APIs are currently implemented. For direct HTTP calls, please refer to the [Prometheus API definition](https://prometheus.io/docs/prometheus/latest/querying/api/#expression-queries): -| Http Method | Path | Prometheus API | Description | -| ----------- | ------------------------------------ | --------------------------------- | -------------------------------- | -| GET/POST | /prom/api/v1/query | /api/v1/query | Query data at a single point in time | -| GET/POST | /prom/api/v1/query_range | /api/v1/query_range | Query data over a time range | -| GET | /prom/api/v1/label/:labelName/values | /api/v1/label//values | Get all labels for a metric | -| GET/POST | /prom/api/v1/series | /api/v1/series | Get all time series | +| Http Method | Path | Prometheus API | Description | +| ----------- | ------------------------------------ | ---------------------------------- | ---------------------------------------- | +| GET/POST | /prom/api/v1/query | /api/v1/query | Query data at a single point in time | +| GET/POST | /prom/api/v1/query_range | /api/v1/query_range | Query data over a time range | +| GET | /prom/api/v1/label/:labelName/values | /api/v1/label//values | Get all label values for a metric | +| GET/POST | /prom/api/v1/series | /api/v1/series | Get all time series | -## Calling Method +## How to Call -You can call the API in DeepFlow as follows: +In DeepFlow, you can call the API as follows: Get the server endpoint port number: @@ -26,7 +26,7 @@ Get the server endpoint port number: port=$(kubectl get --namespace deepflow -o jsonpath="{.spec.ports[0].nodePort}" services deepflow-server) ``` -Example of API call: +Example API calls: Instant Query: @@ -51,38 +51,38 @@ curl -XPOST "http://${deepflow_server_node_ip}:${port}/prom/api/v1/query_range" --data-urlencode "step=60s" ``` -## DeepFlow Metric Definitions +## DeepFlow Metric Definition -When providing PromQL queries externally, DeepFlow metrics are constructed in the format `${database}__${table}__${metric}__${data_precision}`. You can obtain the target data source to query through the definition of [AutoMetrics Metric Types](../../../features/universal-map/auto-metrics/#%E6%8C%87%E6%A0%87%E7%B1%BB%E5%9E%8B). The specific rules are as follows: +When DeepFlow exposes metrics for PromQL queries, the metric name follows the format `${database}__${table}__${metric}__${data_precision}`. You can refer to the [AutoMetrics metric types](../../../features/universal-map/auto-metrics/#%E6%8C%87%E6%A0%87%E7%B1%BB%E5%9E%8B) definition to find the target data source you want to query. The specific rules are as follows: | db | metrics | | ----------------------------------------------------- | ------------------------------------------- | | `flow_log` | `{db}__{table}__{metric}` | -| `flow_metrics` (data_precision values are `1m`/`1s`) | `{db}__{table}__{metric}__{data_precision}` | -| `prometheus` (data written via Prometheus RemoteWrite) | `prometheus__samples__{metric}` | +| `flow_metrics` (data_precision values: `1m`/`1s`) | `{db}__{table}__{metric}__{data_precision}` | +| `prometheus` (data written via Prometheus RemoteWrite)| `prometheus__samples__{metric}` | For example: -- `flow_metrics__application__request__1m`: Represents the number of application layer requests aggregated per minute -- `flow_metrics__network__tcp_timeout__1s`: Represents the number of network layer TCP timeouts aggregated per second -- `flow_log__l7_flow_log__error`: Represents the number of application layer errors +- `flow_metrics__application__request__1m`: queries the number of application-layer requests aggregated per minute +- `flow_metrics__network__tcp_timeout__1s`: queries the number of TCP timeouts at the network layer aggregated per second +- `flow_log__l7_flow_log__error`: queries the number of application-layer errors ## Known Limitations -In the Grafana panel operations, the following limitations are currently known: +In Grafana panel operations, the currently known limitations are: -- Labels cannot be queried directly; you need to select metrics first before choosing labels -- When reselecting metrics, all labels need to be removed first -- Queries will fail if the metrics name contains characters like `.-` that are not supported by Prometheus +- Cannot directly query labels; you must select metrics first before selecting labels +- When reselecting metrics, you must remove all labels first +- Queries will fail if the metric name contains characters like `.-` that are not supported by Prometheus -When querying directly based on PromQL or writing alert rules, the following limitations are currently known: +When directly querying via PromQL or writing alert rules, the currently known limitations are: -- Metrics names cannot be searched using `~/!~` regex -- For metrics provided by DeepFlow, you must first determine the aggregation evaluation method through the [aggregation operator](https://prometheus.io/docs/prometheus/latest/querying/operators/#aggregation-operators) before performing specific metric queries. The functions `stdvar`, `topk`, `bottomk`, and `quantile` are not yet supported and will be supported in future iterations. +- Cannot use `~/!~` regex to search metric names +- For metrics provided by DeepFlow, you must first determine the aggregation evaluation method using an [aggregation operator](https://prometheus.io/docs/prometheus/latest/querying/operators/#aggregation-operators) before performing the specific metric query. The functions `stdvar`, `topk`, `bottomk`, and `quantile` are not yet supported and will be added in future iterations. -# Querying DeepFlow Metrics Based on PromQL +# Querying DeepFlow Metrics with PromQL -Based on the above definitions, we can query DeepFlow metrics using PromQL. Note to `determine the aggregation evaluation method first`, for example: +Based on the above definitions, we can query DeepFlow metrics using PromQL. Note: `determine the aggregation evaluation method first`, for example: - Query the time series trend of all requests with HTTP `500` response codes: @@ -96,13 +96,13 @@ sum(flow_log__l7_flow_log__server_error{response_code="500"}) by(request_resourc sum(flow_log__l7_flow_log__server_error{response_code="500"}) by(auto_service_1, request_resource, response_code) ``` -- Query the trend of TCP connection delay changes over the past 5 minutes with a 10s evaluation interval, grouped by service: +- With a 10s evaluation interval, query the TCP connection latency trend over the past 5 minutes grouped by service: ``` rate(sum(flow_metrics__network_map__rtt__1s)by(auto_service_1)[5m:10s]) ``` -- Query the trend of average application delay over the past 10 minutes with a 1m evaluation interval, grouped by service: +- With a 1m evaluation interval, query the average application latency trend over the past 10 minutes grouped by service: ``` avg_over_time(avg(flow_metrics__application__rrt__1m)by(auto_service)[10m:1m]) @@ -110,9 +110,9 @@ avg_over_time(avg(flow_metrics__application__rrt__1m)by(auto_service)[10m:1m]) # Implementing Prometheus Alerts Based on DeepFlow Metrics -With the above examples, after configuring [Prometheus RemoteRead](../../input/metrics/prometheus/#%E9%85%8D%E7%BD%AE-remote-read), you can build alert rules on Prometheus based on these metrics, such as: +With the above examples, after configuring [Prometheus RemoteRead](../../input/metrics/prometheus/#%E9%85%8D%E7%BD%AE-remote-read), you can build alert rules in Prometheus based on these metrics, for example: -- Alert for requests with latency > 1s and lasting more than 1m: +- Alert for requests with latency > 1s lasting more than 1 minute: ```yaml groups: @@ -126,4 +126,4 @@ groups: description: '{{ $labels.auto_instance }} request to {{ $labels.auto_service }} has a high request latency above 1s (current value: {{ $value }}s)' ``` -We will also support direct configuration of alert rules in future iterations of DeepFlow, so stay tuned. \ No newline at end of file +We will also support directly configuring alert rules in future DeepFlow iterations, so stay tuned. \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/01-query/03-trace-completion.md b/translate/translated/08-integration/03-output/01-query/03-trace-completion.md index 2a2d09e5..e2b75772 100644 --- a/translate/translated/08-integration/03-output/01-query/03-trace-completion.md +++ b/translate/translated/08-integration/03-output/01-query/03-trace-completion.md @@ -7,21 +7,21 @@ permalink: /integration/output/query/trace-completion # Introduction -APM focuses on the code level and lacks the ability to view issues from a full-stack, multi-dimensional perspective without blind spots. Additionally, due to the hindrance of instrumentation, it often fails to cover all services. DeepFlow relies on eBPF zero-instrumentation to fully capture distributed tracing data and generate call chains. In scenarios where DeepFlow and APM are deployed independently, they can collaborate in a loosely coupled manner by using DeepFlow's Trace Completion API to enhance APM's call chains, eliminating blind spots in APM for cloud-native infrastructure and non-instrumented services, significantly reducing the time for triage. +APM focuses on the code level and lacks the capability to observe issues across the full stack and multiple dimensions without blind spots. Additionally, due to the limitations of instrumentation, it is often difficult to cover all services. DeepFlow relies on eBPF for zero-instrumentation, full-coverage collection of distributed tracing data, and correlates it to generate call chains. In scenarios where DeepFlow and APM are deployed completely independently, the two can collaborate in a loosely coupled manner — by using DeepFlow’s Trace Completion API to enhance APM’s call chains, eliminating blind spots in APM for cloud-native infrastructure and non-instrumented services, and significantly shortening the time for problem triage. -Before introducing the API, let's use a diagram to explain the data that APM can complete after calling the DeepFlow API. +Before introducing the API, let’s first use a diagram to illustrate the data that APM can complete after calling the DeepFlow API. ![Full Stack Distributed Tracing](https://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/pub/pic/20230606647ea8bc946f1.jpg) -- In the diagram, Spans starting with A represent application Spans (from APM); those starting with S represent system Spans (from DeepFlow); and those starting with N represent network Spans (from DeepFlow). -- The black parts in the diagram are the input parameters for APM calling the DeepFlow API. DeepFlow will use these `application Spans` as search boundaries to complete the surrounding `system/network Spans` and reconstruct the Parent-Child relationships. -- The blue parts in the diagram are `application Spans` injected with TraceID/SpanID in the protocol from APM, and the `system/network Spans` calculated based on them. These complete the kernel system calls and network transmission paths such as Syscall, Bridge, and IPVS between two services for APM. -- The green parts in the diagram are basic service calls automatically traced by DeepFlow's `system Spans`, such as non-instrumented DNS calls and MySQL calls, Redis calls, etc., where TraceID/SpanID cannot be injected. -- The red parts in the diagram are upstream and downstream services automatically traced by DeepFlow's `system Spans`, such as non-instrumented ALB, NLB, Ingress gateway services, and other services in the business logic that APM has not instrumented. +- Spans starting with **A** in the diagram represent application spans (from APM); those starting with **S** represent system spans (from DeepFlow); those starting with **N** represent network spans (from DeepFlow). +- The black parts in the diagram are the input parameters when APM calls the DeepFlow API. DeepFlow will use these `application spans` as the search boundary, complete the surrounding `system/network spans`, and reconstruct the parent-child relationships. +- The blue parts in the diagram are `system/network spans` calculated based on `application spans` in APM that have TraceID/SpanID injected into the protocol. These fill in kernel system calls such as Syscall, Bridge, IPVS, and network transmission paths between two services for APM. +- The green parts in the diagram are basic service calls automatically traced from DeepFlow’s `system spans`, such as non-instrumented DNS calls, and MySQL/Redis calls where TraceID/SpanID cannot be injected. +- The red parts in the diagram are upstream and downstream services without instrumentation, automatically traced from DeepFlow’s `system spans`, such as non-instrumented ALB, NLB, Ingress gateway services, as well as other services in business logic that APM has not instrumented. # API Description -Get the DeepFlow service endpoint port number: +Get the DeepFlow server endpoint port number: ```bash port=$(kubectl get --namespace deepflow -o jsonpath="{.spec.ports[0].nodePort}" services deepflow-app) @@ -33,7 +33,7 @@ Trace Completion API call method: curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/stats/querier/tracing-completion-by-external-app-spans" ``` -## Input Parameters Description +## Request Parameters ```json { @@ -53,30 +53,30 @@ curl -XPOST "http://${deepflow_server_node_ip}:${port}/v1/stats/querier/tracing- } ``` -| Field | Type | Required | Description | -| ---------------- | --------------- | -------- | ------------------------------------------------------------------------------------------- | -| max_iteration | int | No | Depth of system Span tracing, default is 30, unit: layers | -| network_delay_us | int | No | Time span for network Span tracing, default is 3000000, unit: microseconds | -| app_spans | array[AppSpans] | Yes | List of `application Spans` to complete the call chain, can be all `application Spans` in a complete Trace (not recommended) | +| Field | Type | Required | Description | +| ---------------- | --------------- | -------- | -------------------------------------------------------------------------------------------- | +| max_iteration | int | No | Depth of system span tracing, default 30, unit: layers | +| network_delay_us | int | No | Time span for network span tracing, default 3000000, unit: microseconds | +| app_spans | array[AppSpans] | Yes | List of `application spans` to complete the call chain. Can be all `application spans` in a complete trace (not recommended) | -app_spans are usually part of the application Spans of a Trace in APM. DeepFlow completes based on this. It is recommended to carry the following Spans for each call: +`app_spans` are usually a subset of application spans from a trace in APM. DeepFlow uses them for completion. It is recommended that each call carries the following spans: -- The most concerned application Span (hereinafter referred to as X), and the service it belongs to is called a -- The ancestor Spans of X, until the first ancestor Span that is not service a is found, for example, in SkyWalking, it is the first ancestor Span of type Exit -- The descendant Spans of X, each branch until the first descendant Span that is not service a is found, for example, in SkyWalking, it is the first descendant Span of type Entry for each branch +- The most concerned application span (hereinafter referred to as X), and the service it belongs to is called **a**. +- Ancestor spans of X, until the first ancestor span that does not belong to service **a** is found. For example, in SkyWalking, this is the first ancestor span of type Exit. +- Descendant spans of X, for each sub-branch until the first descendant span that does not belong to service **a** is found. For example, in SkyWalking, this is the first descendant span of type Entry for each sub-branch. -The purpose of carrying these Spans in the request is to inform DeepFlow to complete around Span X and reconstruct the parent-child relationships of all Spans in the returned result with the ancestors and descendants of X as boundaries. The specific parameters required for each app_span are as follows: +The purpose of carrying these spans in the request is to tell DeepFlow to complete the call chain centered on span X, and to reconstruct the parent-child relationships of all spans in the returned result using X’s ancestors and descendants as boundaries. The parameters required for each `app_span` are as follows: -| Field | Type | Required | Description | -| -------------- | ------ | -------- | ------------------------------------------------------------------------------------------------------------------------------------------ | -| trace_id | string | Yes | TraceID of the `application Span` | -| span_id | string | Yes | SpanID of the `application Span` | -| parent_span_id | string | Yes | ParentSpanID of the `application Span` | -| span_kind | int | Yes | Span type of the `application Span`, same meaning as in OpenTelemetry, optional values: 0: unspecified, 1: internal, 2: server, 3: client, 4: producer, 5: consumer | -| start_time_us | int | Yes | Start time of the `application Span`, unit: microseconds | -| end_time_us | int | Yes | End time of the `application Span`, unit: microseconds | +| Field | Type | Required | Description | +| --------------- | ------ | -------- | --------------------------------------------------------------------------------------------------------------------------------- | +| trace_id | string | Yes | TraceID of the `application span` | +| span_id | string | Yes | SpanID of the `application span` | +| parent_span_id | string | Yes | ParentSpanID of the `application span` | +| span_kind | int | Yes | Span type of the `application span`, same meaning as in OpenTelemetry. Options: 0: unspecified, 1: internal, 2: server, 3: client, 4: producer, 5: consumer | +| start_time_us | int | Yes | Start time of the `application span`, unit: microseconds | +| end_time_us | int | Yes | End time of the `application span`, unit: microseconds | -## Output Parameters Description +## Response Parameters ```json { @@ -126,49 +126,49 @@ The purpose of carrying these Spans in the request is to inform DeepFlow to comp } ``` -The tracing in the returned result is the complete Spans traced by DeepFlow, which is an array. Each item in the array is a Span, including both application Spans from APM and system/network Spans from DeepFlow. Important attributes of each Span are: - -| Field | Type | Description | -| ----------------------- | ------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| start_time_us | int | Start time of the Span, unit: microseconds | -| end_time_us | int | End time of the Span, unit: microseconds | -| duration | int | Execution time of the Span, unit: microseconds | -| name | string | Name of the Span, system/network Spans correspond to DeepFlow's [`request_resource` field description](../../../features/universal-map/request-log/) | -| signal_source | int | Source of the Span, corresponding to DeepFlow's [`signal_source` field description](../../../features/universal-map/request-log/) | -| tap_side | int | Span statistics location, corresponding to DeepFlow's [`tap_side` field description](../../../features/universal-map/auto-metrics/#%E7%BB%9F%E8%AE%A1%E4%BD%8D%E7%BD%AE%E8%AF%B4%E6%98%8E) | -| trace_id | string | TraceID, if `system/network Span` has a corresponding `application Span`, it is the value of the corresponding `application Span`; otherwise, the value is empty | -| span_id | string | Original Span ID, if `system/network Span` has a corresponding `application Span`, it is the value of the corresponding `application Span`; otherwise, the value is empty | -| parent_span_id | string | Original parent Span ID, if `system/network Span` has a corresponding `application Span`, it is the value of the corresponding `application Span`; otherwise, the value is empty | -| deepflow_span_id | string | Span ID recalculated by DeepFlow | -| deepflow_parent_span_id | string | Parent Span ID recalculated by DeepFlow | - -In addition, the API will return some extra fields for each Span: - -| Field | Type | Description | Remarks | -| ------------------------- | ------ | ------------------------------------------------------------------------------------------------------------------- | ------------ | -| \_ids | array | DeepFlow call logs corresponding to the Span | | -| related_ids | int | Other DeepFlow call logs related to the Span | | -| flow_id | string | DeepFlow flow logs corresponding to the Span, no data for application/system Spans | -| l7_protocol | int | Application protocol of the Span, corresponding to DeepFlow's [`l7_protocol` field description](../../../features/universal-map/request-log/) | -| l7_protocol_str | string | Application protocol of the Span | -| request_type | string | Request type of the Span | -| request_id | string | Request ID of the Span | -| endpoint | string | Request endpoint of the Span | -| request_resource | string | Request resource of the Span | -| response_status | int | Response status of the Span, corresponding to DeepFlow's [`response_status` field description](../../../features/universal-map/request-log/) | -| process_id | int | Process ID to which the Span belongs, only system Spans have data | -| app_service | string | Service to which the Span belongs, only application Spans have data | -| app_instance | string | Instance to which the Span belongs, only application Spans have data | -| vtap_id | int | Collector ID corresponding to the Span | -| req_tcp_seq | int | TCP Seq corresponding to the Span request, only system/network Spans have data | Used for tracing calculation | -| resp_tcp_seq | int | TCP Seq corresponding to the Span response, only system/network Spans have data | Used for tracing calculation | -| x_request_id | string | X-Request-ID of the Span request or response, only system/network Spans have data | Used for tracing calculation | -| syscall_trace_id_request | string | Syscall TraceID corresponding to the Span request, only system Spans have data | Used for tracing calculation | -| syscall_trace_id_response | string | Syscall TraceID corresponding to the Span response, only system Spans have data | Used for tracing calculation | -| syscall_cap_seq_0 | string | Syscall Seq corresponding to the Span request, only system Spans have data | Used for tracing calculation | -| syscall_cap_seq_1 | string | Syscall Seq corresponding to the Span response, only system Spans have data | Used for tracing calculation | +The `tracing` in the returned result is an array of complete spans traced by DeepFlow. Each item in the array is a span, including both application spans from APM and system/network spans from DeepFlow. Key attributes of each span include: + +| Field | Type | Description | +| ----------------------- | ------ | --------------------------------------------------------------------------------------------------------------------------------------------- | +| start_time_us | int | Span start time, unit: microseconds | +| end_time_us | int | Span end time, unit: microseconds | +| duration | int | Span execution time, unit: microseconds | +| name | string | Span name. For system/network spans, corresponds to DeepFlow’s [`request_resource` field description](../../../features/universal-map/request-log/) | +| signal_source | int | Span source, corresponds to DeepFlow’s [`signal_source` field description](../../../features/universal-map/request-log/) | +| tap_side | int | Span collection location, corresponds to DeepFlow’s [`tap_side` field description](../../../features/universal-map/auto-metrics/#%E7%BB%9F%E8%AE%A1%E4%BD%8D%E7%BD%AE%E8%AF%B4%E6%98%8E) | +| trace_id | string | TraceID. For `system/network spans` with a corresponding `application span`, this is the value from the application span; otherwise empty | +| span_id | string | Original Span ID. For `system/network spans` with a corresponding `application span`, this is the value from the application span; otherwise empty | +| parent_span_id | string | Original Parent Span ID. For `system/network spans` with a corresponding `application span`, this is the value from the application span; otherwise empty | +| deepflow_span_id | string | Span ID recalculated by DeepFlow | +| deepflow_parent_span_id | string | Parent Span ID recalculated by DeepFlow | + +In addition, the API returns extra fields for each span: + +| Field | Type | Description | Notes | +| ------------------------- | ------ | -------------------------------------------------------------------------------------------------------------- | ------------- | +| \_ids | array | DeepFlow request logs corresponding to the span | | +| related_ids | int | Other DeepFlow request logs related to the span | | +| flow_id | string | DeepFlow flow log corresponding to the span. No data for application/system spans | | +| l7_protocol | int | Application protocol of the span, corresponds to DeepFlow’s [`l7_protocol` field description](../../../features/universal-map/request-log/) | +| l7_protocol_str | string | Application protocol of the span | +| request_type | string | Request type of the span | +| request_id | string | Request ID of the span | +| endpoint | string | Request endpoint of the span | +| request_resource | string | Request resource of the span | +| response_status | int | Response status of the span, corresponds to DeepFlow’s [`response_status` field description](../../../features/universal-map/request-log/) | +| process_id | int | Process ID the span belongs to, only available for system spans | +| app_service | string | Service the span belongs to, only available for application spans | +| app_instance | string | Instance the span belongs to, only available for application spans | +| vtap_id | int | Collector ID corresponding to the span | +| req_tcp_seq | int | TCP Seq of the span’s request, only available for system/network spans | For tracing calculation | +| resp_tcp_seq | int | TCP Seq of the span’s response, only available for system/network spans | For tracing calculation | +| x_request_id | string | X-Request-ID of the span’s request or response, only available for system/network spans | For tracing calculation | +| syscall_trace_id_request | string | Syscall TraceID of the span’s request, only available for system spans | For tracing calculation | +| syscall_trace_id_response | string | Syscall TraceID of the span’s response, only available for system spans | For tracing calculation | +| syscall_cap_seq_0 | string | Syscall Seq of the span’s request, only available for system spans | For tracing calculation | +| syscall_cap_seq_1 | string | Syscall Seq of the span’s response, only available for system spans | For tracing calculation | Note: -- The new parent-child relationships of Spans in the returned result need to be constructed using the `deepflow_span_id` and `deepflow_parent_span_id` fields. -- TraceID/SpanID injected into the protocol after application instrumentation can be automatically parsed and collected by the Agent. By default, it is adapted to the Header format of OpenTelemetry and SkyWalking. If there are custom Headers, please modify the Agent configuration. For details, refer to [Agent Advanced Configuration](../../../best-practice/agent-advanced-config/). \ No newline at end of file +- The new parent-child relationships of spans in the returned result should be constructed using the `deepflow_span_id` and `deepflow_parent_span_id` fields. +- TraceID/SpanID injected into the protocol after application instrumentation can be automatically parsed and collected by the Agent. By default, the OpenTelemetry and SkyWalking header formats are supported. For custom headers, please modify the Agent configuration. See [Agent Advanced Configuration](../../../best-practice/agent-advanced-config/) for details. \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/01-query/04-mcp-server.md b/translate/translated/08-integration/03-output/01-query/04-mcp-server.md new file mode 100644 index 00000000..2c1a421b --- /dev/null +++ b/translate/translated/08-integration/03-output/01-query/04-mcp-server.md @@ -0,0 +1,62 @@ +--- +title: MCP Server +permalink: /integration/output/query/mcp-server +--- + +> This document was translated by ChatGPT + +# First eBPF MCP Server Officially Released + +Based on eBPF technology, DeepFlow provides zero-code-intrusion full-stack observability data for cloud-native applications, covering core features such as the universal service map, distributed tracing, and continuous profiling. With the rapid development of AI agent technology and its deep integration into developer workflows, observability tools are facing the challenge of integrating with the AI ecosystem. We are officially releasing the eBPF MCP Server (https://github.com/deepflowio/deepflow/tree/main/server/mcp). The current version focuses on providing continuous profiling capabilities, enabling various AI agents to directly obtain fine-grained, function-level performance analysis results. + +> Model Context Protocol (MCP) is an open protocol launched by Anthropic, designed specifically for standardized interaction between AI models and external tools. MCP defines a unified interface specification with significant advantages: standardized interfaces, ecosystem interoperability, and inherent scalability. + +![eBPF MCP Server](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20250626114123.png) + +# Hands-on Demo - Performance Analysis in AI Coding + +We demonstrate the practical application of DeepFlow MCP Server in an AI-augmented development environment by building a Go application with typical performance bottlenecks. This demo simulates common performance degradation issues in production environments and shows how Cursor can directly call DeepFlow MCP Server via the MCP protocol to obtain continuous profiling results. The entire technical workflow covers the complete analysis chain — from detecting performance anomalies, identifying hotspot functions, correlating with code changes in a specific commit for root cause analysis, to finally outputting actionable, code-level optimization suggestions. + +[AI Coding Hands-on Demo](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/mov/586ca897a5b80c0f443dde84a99f0c99_20250626164636.mov) + +# Get Started Now - Complete Cursor Integration Guide + +> The integration guide uses a K8s environment as an example. + +**Step 1 Deploy DeepFlow in the application runtime environment** + +Deploy DeepFlow (refer to the [deployment documentation](https://deepflow.io/docs/zh/ce-install/overview/)) and enable the continuous profiling feature (refer to the [configuration documentation](https://deepflow.io/docs/zh/features/continuous-profiling/configuration/)). + +![DeepFlow Continuous Profiling](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20250626115101.png) + +**Step 2 Inject the `git_commit_id` label into the Pod** + +To enable performance data queries based on the git commit id, you need to inject the `git_commit_id` label into the application Pod. In production environments, it is recommended to inject it automatically via the CI/CD process. In this tutorial, we use manual YAML modification as an example: + +```yaml +template: + metadata: + labels: + git_commit_id: 7ea306a6dca26d54e65e350439cf8bd0d41c9482 +``` + +**Step 3 Configure DeepFlow MCP Server in Cursor** + +In the `.cursor` folder under the project directory, add a `mcp.json` file with the following content (the default MCP server port is 20080): + +```json +{ + "mcpServers": { + "DeepFlow_Git_Commit_Profile": { + "url": "http://$deepflow_controller_ip:20080/mcp", + "headers": {} + } + } +} +``` + +**Step 4 Open AI Chat in Cursor** + +In Cursor AI Chat, enter the commit id you want to analyze, and the AI will automatically retrieve the performance analysis report for that version. Note: Ensure that the application corresponding to that commit has been deployed in the DeepFlow monitoring environment and that performance profiling data has been collected. + +![Cursor AI Chat](http://yunshan-guangzhou.oss-cn-beijing.aliyuncs.com/yunshan-ticket/png/d2b5ca33bd970f64a6301fa75ae2eb22_20250626115236.png) \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/02-export/01-opentelemetry-exporter.md b/translate/translated/08-integration/03-output/02-export/01-opentelemetry-exporter.md index 201b056c..83306872 100644 --- a/translate/translated/08-integration/03-output/02-export/01-opentelemetry-exporter.md +++ b/translate/translated/08-integration/03-output/02-export/01-opentelemetry-exporter.md @@ -5,21 +5,21 @@ permalink: /integration/output/export/opentelemetry-exporter > This document was translated by ChatGPT -# Features +# Functionality -By converting the standard OTLP protocol, DeepFlow's Span data can be delivered to external platforms, allowing external teams to supplement and enhance their own observability platforms. +After converting to the standard OTLP protocol, DeepFlow's Span data is delivered to external platforms, allowing external teams to supplement and enhance their own observability platforms. -# Span Overview +# Span Introduction In DeepFlow, Spans can be categorized as: - Application Span: Application-level Span data generated using process-level Trace frameworks (Agent/SDK), including custom application Spans, middleware client embedded Spans, communication frameworks, etc. The Trace frameworks here include but are not limited to: Apache SkyWalking Agent, OpenTelemetry Java Agent, and others. -- System Span: Spans collected by DeepFlow through eBPF with zero intrusion, covering system calls, application functions (such as HTTPS), API Gateway, and service mesh Sidecar. +- System Span: Spans collected by DeepFlow through eBPF with zero intrusion, covering system calls, application functions (such as HTTPS), API Gateway, service mesh Sidecar. - Network Span: Spans collected by DeepFlow from network traffic using BPF, covering container network components like iptables/ipvs/OvS/LinuxBridge. # OTel Related -The [OTLP Proto](https://github.com/open-telemetry/opentelemetry-proto/blob/v1.3.1/opentelemetry/proto/trace/v1/trace.proto) can be found here, and the [Trace Semantic Conventions](https://github.com/open-telemetry/semantic-conventions/blob/v1.25.0/docs/general/trace.md) can be seen here. The internal [Resource Semantic Conventions](https://github.com/open-telemetry/semantic-conventions/tree/v1.25.0/docs/resource) can be found here. +Information about [OTLP Proto](https://github.com/open-telemetry/opentelemetry-proto/blob/v1.3.1/opentelemetry/proto/trace/v1/trace.proto) can be found here, and the [Trace Semantic Conventions](https://github.com/open-telemetry/semantic-conventions/blob/v1.25.0/docs/general/trace.md) can be seen here. The [Resource Semantic Conventions](https://github.com/open-telemetry/semantic-conventions/tree/v1.25.0/docs/resource) within Trace can be found here. # Configuration @@ -51,14 +51,14 @@ ingester: # Detailed Parameter Description -| Field | Type | Required | Description | -| ------------- | ------- | -------- | ------------------------------------------------------------------------------------------------------ | -| protocol | string | Yes | Fixed value `opentelemetry` | -| data-sources | strings | Yes | Only supports `flow_log.l7_flow_log` | -| endpoints | strings | Yes | Remote receiving address, only supports gRPC protocol, randomly selects one that can send successfully | -| batch-size | int | No | Batch size, sends in batches when this value is reached. Default: 32 | -| extra-headers | map | No | Header fields for remote gRPC requests, such as tokens for authentication, can be added here | -| export-fields | strings | Yes | Recommended configuration: [$tag, $metrics, $k8s.label] | +| Field | Type | Required | Description | +| ------------- | ------ | -------- | --------------------------------------------------------------------------- | +| protocol | string | Yes | Fixed value `opentelemetry` | +| data-sources | strings| Yes | Only supports `flow_log.l7_flow_log` | +| endpoints | strings| Yes | Remote receiving address, only supports gRPC protocol, randomly selects one that can send successfully | +| batch-size | int | No | Batch size, when this value is reached, it sends in batches. Default: 32 | +| extra-headers | map | No | Header fields for remote gRPC requests, such as tokens for authentication needs, can be supplemented here | +| export-fields | strings| Yes | Recommended configuration: [$tag, $metrics, $k8s.label] | [Detailed Configuration Reference](./exporter-config/) @@ -68,192 +68,192 @@ In Flow_log, there is an internal logic that categorizes all data hierarchically ### Tracing Info -Span-related data belonging to OTel remains unchanged, while other fields are included in span.attributes. +Data related to OTel's Span remains unchanged, while other fields are included in span.attributes. -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------------ | :-------------- | :------------------------------- | :------ | -| x_request_id | span.attributes | df.span.x_request_id | | -| syscall_trace_id_request | span.attributes | df.span.syscall_trace_id_request | | -| syscall_trace_id_response | span.attributes | df.span.syscall_thread_0 | | -| syscall_thread_0 | span.attributes | df.span.syscall_thread_0 | | -| syscall_thread_1 | span.attributes | df.span.syscall_thread_1 | | -| syscall_cap_seq_0 | span.attributes | df.span.syscall_cap_seq_0 | | -| syscall_cap_seq_1 | span.attributes | df.span.syscall_cap_seq_1 | | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :-------------------------- | :---------------- | :------------------------------ | :------ | +| x_request_id | span.attributes | df.span.x_request_id | | +| syscall_trace_id_request | span.attributes | df.span.syscall_trace_id_request| | +| syscall_trace_id_response | span.attributes | df.span.syscall_thread_0 | | +| syscall_thread_0 | span.attributes | df.span.syscall_thread_0 | | +| syscall_thread_1 | span.attributes | df.span.syscall_thread_1 | | +| syscall_cap_seq_0 | span.attributes | df.span.syscall_cap_seq_0 | | +| syscall_cap_seq_1 | span.attributes | df.span.syscall_cap_seq_1 | Remarks:| ### Service Info -Service application-level information, all included in resource.attributes, including application-related, process, and thread-related information. For special requirements regarding process and thread-related information, please use OTel Processor for conversion. +Service application-level information is all included in resource.attributes, including application-related, process, and thread-related information. For special needs regarding process and thread, please use OTel Processor for conversion. -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :------------------ | :------------- | -| auto_service | resource.attributes | service.name | Standard field | -| auto_instance | resource.attributes | service.instance.id | Standard | -| process_id | resource.attributes | process.pid | | -| process_kname | resource.attributes | thread.name | | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :------------------ | :--------------------- | :----------------- | :------ | +| auto_service | resource.attributes | service.name | Standard field | +| auto_instance | resource.attributes | service.instance.id| Standard | +| process_id | resource.attributes | process.pid | Remarks: | +| process_kname | resource.attributes | thread.name | Remarks: | ### Flow Info -Including fields: \_id, time, flow_id, start_time, end_time, close_type, status, is_new_flow. +Includes fields: \_id, time, flow_id, start_time, end_time, close_type, status, is_new_flow. -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------------ | :------------------------ | :-------------------------------------------------------------------- | -| \_id | resource.attributes | df.flow_info.id | | -| time | resource.attributes | df.flow_info.time | | -| flow_id | resource.attributes | df.flow_info.flow_id | | -| start_time | span.start_time_unix_nano | span.start_time_unix_nano | Note time format conversion, will be converted to OTel-compliant time | -| end_time | span.end_time_unix_nano | span.end_time_unix_nano | Note time format conversion, will be converted to OTel-compliant time | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :------------------ | :------------------------- | :----------------------- | :---------------------------------------- | +| \_id | resource.attributes | df.flow_info.id | | +| time | resource.attributes | df.flow_info.time | | +| flow_id | resource.attributes | df.flow_info.flow_id | | +| start_time | span.start_time_unix_nano | span.start_time_unix_nano| Note time format conversion, will convert to OTel-compliant time | +| end_time | span.end_time_unix_nano | span.end_time_unix_nano | Note time format conversion, will convert to OTel-compliant time | ### Capture Info -Including fields: signal_source, agent, nat_source, capture_nic, capture_nic_name, capture_nic_type, observation_point, l2_end, l3_end, has_pcap, nat_real_ip, nat_real_port +Includes fields: signal_source, agent, nat_source, capture_nic, capture_nic_name, capture_nic_type, observation_point, l2_end, l3_end, has_pcap, nat_real_ip, nat_real_port -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :-------------------------------- | :------ | -| signal_source | resource.attributes | df.capture_info.signal_source | | -| agent | resource.attributes | df.capture_info.agent | | -| nat_source | resource.attributes | df.capture_info.nat_source | | -| capture_nic | resource.attributes | df.capture_info.capture_nic | | -| capture_nic_name | resource.attributes | df.capture_info.capture_nic_name | | -| capture_nic_type | resource.attributes | df.capture_info.capture_nic_type | | -| observation_point | resource.attributes | df.capture_info.observation_point | | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :------------------ | :--------------------- | :------------------------------- | :------ | +| signal_source | resource.attributes | df.capture_info.signal_source | | +| agent | resource.attributes | df.capture_info.agent | | +| nat_source | resource.attributes | df.capture_info.nat_source | | +| capture_nic | resource.attributes | df.capture_info.capture_nic | | +| capture_nic_name | resource.attributes | df.capture_info.capture_nic_name | | +| capture_nic_type | resource.attributes | df.capture_info.capture_nic_type | | +| observation_point | resource.attributes | df.capture_info.observation_point| | ### Universal Tag -Including fields: region, az, host, chost, vpc, l2_vpc, subnet, router, dhcpgw, lb, lb_listener, natgw, pod_cluster, pod_ns, pod_node, pod_ingress, pod_service, pod_group, pod, service, auto_service, auto_service_type, auto_instance, auto_instance_type - -For necessary conversions, please use OTel Processor. - -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :---------------------------------- | :----------------------------------------------------------------------------------------------------------- | -| region | resource.attributes | df.universal_tag.region | | -| az | resource.attributes | df.universal_tag.az | | -| host | resource.attributes | df.universal_tag.host | | -| chost | resource.attributes | df.universal_tag.chost | | -| vpc | resource.attributes | df.universal_tag.vpc | | -| l2_vpc | resource.attributes | df.universal_tag.l2_vpc | | -| subnet | resource.attributes | df.universal_tag.subnet | | -| router | resource.attributes | df.universal_tag.router | | -| dhcpgw | resource.attributes | df.universal_tag.dhcpgw | | -| lb | resource.attributes | df.universal_tag.lb | | -| lb_listener | resource.attributes | df.universal_tag.lb_listener | | -| natgw | resource.attributes | df.universal_tag.natgw | | -| pod_cluster | resource.attributes | df.universal_tag.pod_cluster | According to the official documentation, this should be k8s.pod_cluster, convert if necessary | -| pod_ns | resource.attributes | df.universal_tag.pod_ns | According to the official documentation, this should be k8s.pod_ns, convert if necessary | -| pod_node | resource.attributes | df.universal_tag.pod_node | According to the official documentation, this should be k8s.pod_node, convert if necessary | -| pod_ingress | resource.attributes | df.universal_tag.pod_ingress | According to the official documentation, this should be k8s.pod_ingress, convert if necessary | -| pod_service | resource.attributes | df.universal_tag.pod_service | According to the official documentation, this should be k8s.pod_service, convert if necessary | -| pod_group | resource.attributes | df.universal_tag.pod_group | According to the official documentation, this should be k8s.pod_group, convert if necessary | -| pod | resource.attributes | df.universal_tag.pod | According to the official documentation, this should be k8s.pod.xxx, semantics unclear, convert if necessary | -| pod_cluster | resource.attributes | df.universal_tag.pod_cluster | According to the official documentation, this should be k8s.pod_cluster, convert if necessary | -| service | resource.attributes | df.universal_tag.service | | -| auto_service | resource.attributes | df.universal_tag.auto_service | | -| auto_service_type | resource.attributes | df.universal_tag.auto_service_type | | -| auto_instance | resource.attributes | df.universal_tag.auto_instance | | -| auto_instance_type | resource.attributes | df.universal_tag.auto_instance_type | | +Includes fields: region, az, host, chost, vpc, l2_vpc, subnet, router, dhcpgw, lb, lb_listener, natgw, pod_cluster, pod_ns, pod_node, pod_ingress, pod_service, pod_group, pod, service, auto_service, auto_service_type, auto_instance, auto_instance_type + +If needed, please use OTel Processor for format conversion. + +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :-------------------- | :--------------------- | :--------------------------------- | :------------------------------------------------------------------- | +| region | resource.attributes | df.universal_tag.region | | +| az | resource.attributes | df.universal_tag.az | | +| host | resource.attributes | df.universal_tag.host | | +| chost | resource.attributes | df.universal_tag.chost | | +| vpc | resource.attributes | df.universal_tag.vpc | | +| l2_vpc | resource.attributes | df.universal_tag.l2_vpc | | +| subnet | resource.attributes | df.universal_tag.subnet | | +| router | resource.attributes | df.universal_tag.router | | +| dhcpgw | resource.attributes | df.universal_tag.dhcpgw | | +| lb | resource.attributes | df.universal_tag.lb | | +| lb_listener | resource.attributes | df.universal_tag.lb_listener | | +| natgw | resource.attributes | df.universal_tag.natgw | | +| pod_cluster | resource.attributes | df.universal_tag.pod_cluster | According to official documentation, this should be k8s.pod_cluster, convert if needed | +| pod_ns | resource.attributes | df.universal_tag.pod_ns | According to official documentation, this should be k8s.pod_ns, convert if needed | +| pod_node | resource.attributes | df.universal_tag.pod_node | According to official documentation, this should be k8s.pod_node, convert if needed | +| pod_ingress | resource.attributes | df.universal_tag.pod_ingress | According to official documentation, this should be k8s.pod_ingress, convert if needed | +| pod_service | resource.attributes | df.universal_tag.pod_service | According to official documentation, this should be k8s.pod_service, convert if needed | +| pod_group | resource.attributes | df.universal_tag.pod_group | According to official documentation, this should be k8s.pod_group, convert if needed | +| pod | resource.attributes | df.universal_tag.pod | According to official documentation, this should be k8s.pod.xxx, semantics unclear, convert if needed | +| pod_cluster | resource.attributes | df.universal_tag.pod_cluster | According to official documentation, this should be k8s.pod_cluster, convert if needed | +| service | resource.attributes | df.universal_tag.service | | +| auto_service | resource.attributes | df.universal_tag.auto_service | | +| auto_service_type | resource.attributes | df.universal_tag.auto_service_type | | +| auto_instance | resource.attributes | df.universal_tag.auto_instance | | +| auto_instance_type | resource.attributes | df.universal_tag.auto_instance_type| | ### Custom Tag -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :--------------------------- | :------ | -| k8s.labels.xxx | resource.attributes | df.custom_tag.k8s.labels.xxx | | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :------------------ | :--------------------- | :-------------------------- | :------ | +| k8s.labels.xxx | resource.attributes | df.custom_tag.k8s.labels.xxx| Remarks:| ### Network Layer -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :------------------------------------------ | :------------- | -| ip | resource.attributes | df.network.ip | | -| is_ipv4 | resource.attributes | df.network.is_ipv4 | | -| is_internet | resource.attributes | df.network.is_internet | | -| protocol | resource.attributes | net.transport = ip\_(lowercase ${protocol}) | Standard field | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :------------------ | :-------------------- | :--------------------------------------- | :------ | +| ip | resource.attributes | df.network.ip | | +| is_ipv4 | resource.attributes | df.network.is_ipv4 | | +| is_internet | resource.attributes | df.network.is_internet | | +| protocol | resource.attributes. | net.transport = ip\_(lowercase ${protocol}) | Standard field | ### Transport Layer -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :------------------------------ | :------ | -| client_port | resource.attributes | df.transport.client_port | | -| server_port | resource.attributes | df.transport.server_port | | -| tcp_flags_bit | resource.attributes | df.transport.tcp_flags_bit | | -| syn_seq | resource.attributes | df.transport.syn_seq | | -| syn_ack_seq | resource.attributes | df.transport.syn_ack_seq | | -| last_keepalive_seq | resource.attributes | df.transport.last_keepalive_seq | | -| last_keepalive_ack | resource.attributes | df.transport.last_keepalive_ack | | -| req_tcp_seq | resource.attributes | df.transport.req_tcp_seq | | -| resp_tcp_seq | resource.attributes | df.transport.resp_tcp_seq | | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :--------------------- | :----------------------------- | :------ | +| client_port | resource.attributes | df.transport.client_port | | +| server_port | resource.attributes | df.transport.server_port | | +| tcp_flags_bit | resource.attributes | df.transport.tcp_flags_bit | | +| syn_seq | resource.attributes | df.transport.syn_seq | | +| syn_ack_seq | resource.attributes | df.transport.syn_ack_seq | | +| last_keepalive_seq | resource.attributes | df.transport.last_keepalive_seq| | +| last_keepalive_ack | resource.attributes | df.transport.last_keepalive_ack| Remarks:| +| req_tcp_seq | resource.attributes | df.transport.req_tcp_seq | | +| resp_tcp_seq | resource.attributes | df.transport.resp_tcp_seq | | ### Application Layer -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :------------------------- | :-------------------- | -| l7_protocol | resource.attributes | df.application.l7_protocol | Field mapping details | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :------------------ | :--------------------- | :------------------------ | :-------------- | +| l7_protocol | resource.attributes | df.application.l7_protocol| Field mapping details | # Protocol Field Mapping -Here, special mappings of protocol-specific fields to OTLP standard fields are supplemented (general fields can be found above): +Here, special field mappings for each protocol to OTLP standard fields are supplemented (for general fields, please refer to the above): ## Application Protocol Additional Fields -The following fields apply to all application layer protocols: +The following fields apply to each application layer protocol: -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :------------------ | :--------------------------------------- | :--------------------------------------------------------------------- | -| None | resource.attributes | telemetry.sdk.name=deepflow | Custom | -| None | resource.attributes | telemetry.sdk.version=${current version} | Custom | -| chost_0/pod_node_0 | span.attributes | net.host.name | Standard, first get chost_x, if not present, try to get pod_node_x | -| chost_1/pod_node_1 | span.attributes | net.peer.name | Standard, first get chost_x, if not present, try to get pod_node_x | -| client_port | span.attributes | net.host.port | Standard | -| server_port | span.attributes | net.peer.port | Standard | -| ip_0 | span.attributes | net.sock.host.addr | Standard | -| ip_1 | span.attributes | net.sock.peer.addr | Standard | -| response_status | span.status | span.status | 0: OK -> Ok; server error, client error -> Error; not present -> Unset | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :--------------------- | :------------------------------- | :----------------------------------------------------------- | +| None | resource.attributes | telemetry.sdk.name=deepflow | Custom | +| None | resource.attributes | telemetry.sdk.version=${current version} | Custom | +| chost_0/pod_node_0 | span.attributes | net.host.name | Standard, first get chost_x, if not present, try to get pod_node_x | +| chost_1/pod_node_1 | span.attributes | net.peer.name | Standard, first get chost_x, if not present, try to get pod_node_x | +| client_port | span.attributes | net.host.port | Standard | +| server_port | span.attributes | net.peer.port | Standard | +| ip_0 | span.attributes | net.sock.host.addr | Standard | +| ip_1 | span.attributes | net.sock.peer.addr | Standard | +| response_status | span.status | span.status | 0: Normal -> Ok 1; Server exception, client exception -> Error; Not present -> Unset | -## HTTP Protocol Suite +## HTTP Protocol Cluster ### HTTP -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :--------------------------------------------- | :------------------------------- | -| version | span.attributes | http.flavor | Standard field | -| request_type | span.attributes | http.method | Standard field | -| request_domain | span.attributes | net.peer.name | Standard field | -| request_resource | span.attributes | df.http.path | Custom | -| request_id | span.attributes | df.global.request_id | Custom | -| response_code | span.attributes | http.status_code | Standard field | -| response_exception | span.event | event.name | Standard field | -| http_proxy_client | span.attributes | df.http.proxy_client | Custom | -| None | span.name | span.name= ${request_type} + ${request_source} | Standard field, space in between | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :-------------------------------------------- | :--------------- | +| version | span.attributes | http.flavor | Standard field | +| request_type | span.attributes | http.method | Standard field | +| request_domain | span.attributes | net.peer.name | Standard field | +| request_resource | span.attributes | df.http.path | Custom | +| request_id | span.attributes | df.global.request_id | Custom | +| response_code | span.attributes | http.status_code | Standard field | +| response_exception | span.event | event.name | Standard field | +| http_proxy_client | span.attributes | df.http.proxy_client | Custom | +| None | span.name | span.name= ${request_type} + ${request_source}| Standard field with space in between | ### HTTP2 TODO -## RPC Protocol Suite +## RPC Protocol Cluster ### Dubbo -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :------------------------------------------------------------------ | :---------------------------------------------------------- | -| None | span.attributes | rpc.system=apache_dubbo | Standard field | -| None | span.attributes | rpc.service=${request_resource} | Standard field | -| None | span.attributes | rpc.method=${request_type} | Standard field | -| None | span.attributes | span.name= ${request_source} + "/" + ${request_type} == ${endpoint} | Standard field, prioritize concatenation | -| response_exception | span.event | event.name | Standard field | -| request_domain | span.attributes | df.dubbo.request_domain | If not obtainable, use net.peer.name as an additional field | -| version | span.attributes | df.dubbo.version | Custom | -| request_id | span.attributes | df.global.request_id | Custom | -| response_code | span.attributes | df.response_code | Custom | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :----------------------------------------------------------------- | :---------------------------------------- | +| None | span.attributes | rpc.system=apache_dubbo | Standard field | +| None | span.attributes | rpc.service=${request_resource} | Standard field | +| None | span.attributes | rpc.method=${request_type} | Standard field | +| None | span.attributes | span.name= ${request_source} + "/" + ${request_type} == ${endpoint}| Standard field, prioritize concatenation | +| response_exception | span.event | event.name | Standard field | +| request_domain | span.attributes | df.dubbo.request_domain | Cannot be obtained as net.peer.name, use as additional field | +| version | span.attributes | df.dubbo.version | Custom | +| request_id | span.attributes | df.global.request_id | Custom | +| response_code | span.attributes | df.response_code | Custom | ### gRPC -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :------------------------------------------------------------------ | :---------------------------------------------------------- | -| None | span.attributes | rpc.system=grpc | Standard field | -| None | span.attributes | rpc.system=${request_resource} | Standard field | -| None | span.attributes | rpc.system=${request_type} | Standard field | -| None | span.attributes | span.name= ${request_source} + "/" + ${request_type} == ${endpoint} | Standard field | -| response_exception | span.event | event.name | Standard field | -| version | span.attributes | http.flavor | Standard field | -| request_domain | span.attributes | df.grpc.request_domain | If not obtainable, use net.peer.name as an additional field | -| request_id | span.attributes | df.global.request_id | Custom | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :----------------------------------------------------------------- | :---------------------------------------- | +| None | span.attributes | rpc.system=grpc | Standard field | +| None | span.attributes | rpc.system=${request_resource} | Standard field | +| None | span.attributes | rpc.system=${request_type} | Standard field | +| None | span.attributes | span.name= ${request_source} + "/" + ${request_type} == ${endpoint}| Standard field | +| response_exception | span.event | event.name | Standard field | +| version | span.attributes | http.flavor | Standard field | +| request_domain | span.attributes | df.grpc.request_domain | Cannot be obtained as net.peer.name, use as additional field | +| request_id | span.attributes | df.global.request_id | Custom | ### SOFARPC @@ -263,89 +263,89 @@ TODO TODO -## SQL Protocol Suite +## SQL Protocol Cluster ### MySQL -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :-------------------------------------- | :------------------------------------------ | -| None | span.attributes | db.system==mysql | Standard | -| None | span.attributes | db.operation=${C/R/U/D} | Standard field | -| None | span.attributes | db.statement=${request_resource} | Standard field | -| request_type | span.attributes | df.mysql.request_type | Custom: db.operation defined as SQL keyword | -| response_exception | span.event | event.name | Standard field | -| None | span.name | span.name=${C/R/U/D} + ${db} + ${table} | Standard field | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :------------------------------------- | :-------------------------------------- | +| None | span.attributes | db.system==mysql | Standard | +| None | span.attributes | db.operation=${C/R/U/D} | Standard field | +| None | span.attributes | db.statement=${request_resource} | Standard field | +| request_type | span.attributes | df.mysql.request_type | Custom: db.operation defined as SQL keyword | +| response_exception | span.event | event.name | Standard field | +| None | span.name | span.name=${C/R/U/D} + ${db} + ${table}| Standard field | ### PostgreSQL -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :-------------------------------------- | :------------------------------------------ | -| None | span.attributes | db.system==postgresql | Standard | -| None | span.attributes | db.operation=${C/R/U/D} | Standard field | -| None | span.attributes | db.statement=${request_resource} | Standard field | -| request_type | span.attributes | df.postgresql.request_type | Custom: db.operation defined as SQL keyword | -| response_exception | span.event | event.name | Standard field | -| None | span.name | span.name=${C/R/U/D} + ${db} + ${table} | Standard field | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :------------------------------------- | :-------------------------------------- | +| None | span.attributes | db.system==postgresql | Standard | +| None | span.attributes | db.operation=${C/R/U/D} | Standard field | +| None | span.attributes | db.statement=${request_resource} | Standard field | +| request_type | span.attributes | df.postgresql.request_type | Custom: db.operation defined as SQL keyword | +| response_exception | span.event | event.name | Standard field | +| None | span.name | span.name=${C/R/U/D} + ${db} + ${table}| Standard field | -## NoSQL Protocol Suite +## NoSQL Protocol Cluster ### Redis -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :------------------------------- | :------------- | -| None | span.attributes | db.system==redis | Custom | -| None | span.attributes | db.operation=${request_type} | Standard field | -| None | span.attributes | db.statement=${request_resource} | Standard field | -| response_exception | span.event | event.name | Standard field | -| None | span.name | span.name=${request_type} | Standard field | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :------------------------------ | :------ | +| None | span.attributes | db.system==redis | Custom | +| None | span.attributes | db.operation=${request_type} | Standard field | +| None | span.attributes | db.statement=${request_resource}| Standard field | +| response_exception | span.event | event.name | Standard field | +| None | span.name | span.name=${request_type} | Standard field | ### MongoDB -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :------------------------------- | :------------- | -| None | span.attributes | db.system==mongodb | Custom | -| None | span.attributes | db.operation=${request_type} | Standard field | -| None | span.attributes | db.statement=${request_resource} | Standard field | -| response_exception | span.event | event.name | Standard field | -| None | span.name | span.name=${request_type} | Standard field | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :------------------------------ | :------ | +| None | span.attributes | db.system==mongodb | Custom | +| None | span.attributes | db.operation=${request_type} | Standard field | +| None | span.attributes | db.statement=${request_resource}| Standard field | +| response_exception | span.event | event.name | Standard field | +| None | span.name | span.name=${request_type} | Standard field | -## Messaging Protocols +## Message Queue Protocol Cluster ### Kafka -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :------------------------- | :------------- | -| None | span.attributes | messaging.system=kafka | Standard | -| None | span.name | span.name=${request_type} | Non-standard | -| request_type | span.attributes | df.kafka.request_type | Custom | -| request_id | span.attributes | df.global.request_id | Custom | -| request_resource | span.attributes | df.global.request_resource | Custom | -| request_domain | span.attributes | df.kafka.request_domain | Custom | -| response_code | span.attributes | df.kafka.response_code | Custom | -| response_exception | span.event | event.name | Standard field | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :------------------------ | :------ | +| None | span.attributes | messaging.system=kafka | Standard| +| None | span.name | span.name=${request_type} | Non-standard | +| request_type | span.attributes | df.kafka.request_type | Custom | +| request_id | span.attributes | df.global.request_id | Custom | +| request_resource | span.attributes | df.global.request_resource| Custom | +| request_domain | span.attributes | df.kafka.request_domain | Custom | +| response_code | span.attributes | df.kafka.response_code | Custom | +| response_exception | span.event | event.name | Standard field | ### MQTT -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :----------------------------------------------------------- | :---------------------------------------------------------------------- | -| None | span.attributes | messaging.system=mqtt | Standard | -| None | span.attributes | messaging.operation=${request_type} | Standard: PUBLISH -> publish, SUBSCRIBE -> process, others filtered out | -| None | span.name | span.name=${request_resource} + " " + ${messaging.operation} | Standard | -| request_type | span.attributes | df.mqtt.request_type | Custom | -| request_resource | span.attributes | df.mqtt.request_resource | Custom | -| request_domain | span.attributes | df.mqtt.request_domain | Custom | -| response_code | span.attributes | df.mqtt.response_code | Custom | -| response_exception | span.event | event.name | Standard field | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :---------------------------------------------------------- | :--------------------------------------------------------------- | +| None | span.attributes | messaging.system=mqtt | Standard | +| None | span.attributes | messaging.operation=${request_type} | Standard, where: PUBLISH -> publish, SUBSCRIBE -> process, others are filtered out | +| None | span.name | span.name=${request_resource} + " " + ${messaging.operation}| Standard. | +| request_type | span.attributes | df.mqtt.request_type | Custom | +| request_resource | span.attributes | df.mqtt.request_resource | Custom | +| request_domain | span.attributes | df.mqtt.request_domain | Custom | +| response_code | span.attributes | df.mqtt.response_code | Custom | +| response_exception | span.event | event.name | Standard field | -## Network Protocols +## Network Protocol Cluster ### DNS -| Original Field Name | Mapped Location | Mapped Name | Remarks | -| :------------------ | :-------------- | :---------------------- | :------------- | -| request_type | span.attributes | df.dns.request_type | Custom | -| request_resource | span.attributes | df.dns.request_resource | Custom | -| request_id | span.attributes | df.global.request_id | Custom | -| response_code | span.attributes | df.dns.response_code | Custom | -| response_exception | span.event | event.name | Standard field | -| response_result | span.attributes | df.dns.response_result | Custom | +| Original Field Name | Mapped Location | Mapped Name | Remarks | +| :---------------------- | :---------------- | :--------------------- | :------ | +| request_type | span.attributes | df.dns.request_type | Custom | +| request_resource | span.attributes | df.dns.request_resource| Custom | +| request_id | span.attributes | df.global.request_id | Custom | +| response_code | span.attributes | df.dns.response_code | Custom | +| response_exception | span.event | event.name | Standard field | +| response_result | span.attributes | df.dns.response_result | Custom | \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/02-export/02-prom-remote-write.md b/translate/translated/08-integration/03-output/02-export/02-prom-remote-write.md index 438fc6c7..5cfb3550 100644 --- a/translate/translated/08-integration/03-output/02-export/02-prom-remote-write.md +++ b/translate/translated/08-integration/03-output/02-export/02-prom-remote-write.md @@ -5,22 +5,22 @@ permalink: /integration/output/export/prometheus-remote-write > This document was translated by ChatGPT -# Functionality +# Function -Using Prometheus Remote Write, you can export metrics generated by DeepFlow to external platforms. This allows you to continue leveraging the Prometheus ecosystem, such as viewing metrics and configuring alerts through Prometheus. +By using Prometheus Remote Write, you can export the metrics generated by DeepFlow to external platforms. This allows you to continue leveraging the Prometheus ecosystem, such as viewing metrics and configuring alerts through Prometheus. -# Metrics Overview +# Introduction to Metrics -Within DeepFlow, metrics can be categorized into two types: +In DeepFlow, metrics can be categorized into two types: -- Application Performance Metrics: [Refer to details](../../../features/universal-map/application-metrics/) - - Corresponds to `flow_metrics.application*` table data in ClickHouse -- Network Performance Metrics: [Refer to details](../../../features/universal-map/network-metrics/) - - Corresponds to `flow_metrics.network*` table data in ClickHouse +- Application performance metrics: [See details here](../../../features/universal-map/application-metrics/) + - Corresponds to the `flow_metrics.application*` tables in ClickHouse +- Network performance metrics: [See details here](../../../features/universal-map/network-metrics/) + - Corresponds to the `flow_metrics.network*` tables in ClickHouse # Prometheus Remote Write -For protocol format, refer to Prometheus's pb file definition: https://github.com/prometheus/prometheus/blob/main/prompb/remote.proto +For the protocol format, refer to Prometheus’s pb file definition: https://github.com/prometheus/prometheus/blob/main/prompb/remote.proto # DeepFlow Server Configuration Guide @@ -60,22 +60,22 @@ ingester: # Detailed Parameter Description -| Field | Type | Required | Description | -| -------------- | ------- | -------- | ----------------------------------------------------------------------- | -| protocol | string | Yes | Fixed value `prometheus` | -| data-sources | string | Yes | Values from `flow_metrics.*` data, does not support `flow_log.*` data | -| endpoints | string | Yes | Remote receiving address, remote write receiving address, randomly selects one that can send successfully | -| batch-size | int | No | Batch size, sends in batches when this value is reached. Default: 1024 | -| extra-headers | map | No | Header fields for remote HTTP requests, such as tokens for authentication | -| export-fields | string | Yes | Currently does not support `$k8s.label`, recommended configuration: [$tag, $metrics] | +| Field | Type | Required | Description | +| ------------- | ------- | -------- | --------------------------------------------------------------------------- | +| protocol | strings | Yes | Fixed value `prometheus` | +| data-sources | strings | Yes | Use `flow_metrics.*` data, `flow_log.*` and other data types are not supported | +| endpoints | strings | Yes | Remote receiving addresses for remote write; randomly selects one that can send successfully | +| batch-size | int | No | Batch size; when this value is reached, data is sent in batches. Default: 1024 | +| extra-headers | map | No | HTTP header fields for remote requests; for example, tokens for authentication can be added here | +| export-fields | strings | Yes | `$k8s.label` is not supported; recommended configuration: [$tag, $metrics] | -[Refer to detailed configuration](./exporter-config/) +[See detailed configuration reference](./exporter-config/) -# Quick Practice Demo +# Quick Demo -- Set up a RemoteWrite receiver, refer to this Prometheus [demo](https://github.com/prometheus/prometheus/tree/main/documentation/examples/remote_storage/example_write_adapter) +- Set up a RemoteWrite receiver; you can refer to this Prometheus [demo](https://github.com/prometheus/prometheus/tree/main/documentation/examples/remote_storage/example_write_adapter) -- Add configuration +- Add the configuration: ```yaml exporters: @@ -88,6 +88,6 @@ exporters: - $metrics ``` -- Restart DeepFlow Server, and after a short wait, you should see the output results at the RemoteWrite receiver as shown in the image +- Restart the DeepFlow Server, and after a short while, you should see output results like the following on the RemoteWrite receiver: ![](./imgs/remote-write.png) \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/02-export/03-kafka-exporter.md b/translate/translated/08-integration/03-output/02-export/03-kafka-exporter.md index 3700c714..37fd640f 100644 --- a/translate/translated/08-integration/03-output/02-export/03-kafka-exporter.md +++ b/translate/translated/08-integration/03-output/02-export/03-kafka-exporter.md @@ -5,18 +5,18 @@ permalink: /integration/output/export/kafka-exporter > This document was translated by ChatGPT -# Functionality +# Function -Using Kafka, you can export metrics, flow logs, call logs, and IO events generated by DeepFlow to external platforms. +Through Kafka, DeepFlow can export generated metrics, flow logs, call logs, and IO events to external platforms. -# Metrics Overview +# Introduction to Metrics -Within DeepFlow, metrics can be categorized into two types: +In DeepFlow, metrics can be divided into two types: -- Application Performance Metrics: [Refer to details](../../../features/universal-map/application-metrics/) - - Corresponds to `flow_metrics.application*` table data in ClickHouse -- Network Performance Metrics: [Refer to details](../../../features/universal-map/network-metrics/) - - Corresponds to `flow_metrics.network*` table data in ClickHouse +- Application performance metrics: [See details here](../../../features/universal-map/application-metrics/) + - Corresponds to the `flow_metrics.application*` tables in ClickHouse +- Network performance metrics: [See details here](../../../features/universal-map/network-metrics/) + - Corresponds to the `flow_metrics.network*` tables in ClickHouse # Kafka Export @@ -24,7 +24,7 @@ The protocol format uses JSON. # DeepFlow Server Configuration Guide -To enable metrics export, add the following configuration under the Server settings: +Add the following configuration under the Server configuration to enable metrics export: ```yaml ingester: @@ -70,11 +70,11 @@ ingester: | Field | Type | Required | Description | | ------------- | ------- | -------- | --------------------------------------------------------------------------------------------- | | protocol | string | Yes | Fixed value `kafka` | -| data-sources | strings | Yes | Values from ClickHouse `flow_metrics.*/flow_log.*/event.perf_event` data, also used for Kafka topic names | -| endpoints | strings | Yes | Remote receiving addresses, Kafka broker receiving addresses, randomly select one that can send successfully | -| batch-size | int | No | Batch size, when this value is reached, send in batches. Default value: 1024 | +| data-sources | strings | Yes | Values from ClickHouse `flow_metrics.*/flow_log.*/event.perf_event` data, also used as Kafka topic names | +| endpoints | strings | Yes | Remote receiving addresses, Kafka broker addresses; randomly selects one that can send successfully | +| batch-size | int | No | Batch size; when this value is reached, data is sent in batches. Default: 1024 | | export-fields | strings | Yes | Recommended configuration: [$tag, $metrics] | -| sasl | struct | No | Kafka connection authentication method, currently only supports 'SASL_SSL' with 'PLAIN' method | -| topic | string | No | Kafka topic name, if empty, the default value is `deepflow.$data-source`, such as `deepflow.flow_log.l7_flow_log` | +| sasl | struct | No | Kafka connection authentication method; currently only supports 'SASL_SSL' with 'PLAIN' | +| topic | string | No | Kafka topic name; if empty, defaults to `deepflow.$data-source`, e.g., `deepflow.flow_log.l7_flow_log` | -[Refer to detailed configuration](./exporter-config/) \ No newline at end of file +[See detailed configuration reference](./exporter-config/) \ No newline at end of file diff --git a/translate/translated/08-integration/03-output/02-export/04-exporter-config.md b/translate/translated/08-integration/03-output/02-export/04-exporter-config.md index dc82e9ec..746b3cea 100644 --- a/translate/translated/08-integration/03-output/02-export/04-exporter-config.md +++ b/translate/translated/08-integration/03-output/02-export/04-exporter-config.md @@ -52,37 +52,37 @@ ingester: # Detailed Parameter Description -| Field | Type | Required | Description | -| ---------------------------------------- | ------- | -------- | ----------------------------------------------------------------------------------------------------------------------------------- | -| protocol | string | Yes | Supports `opentelemetry`, `prometheus`, `kafka` | -| enabled | bool | Yes | Whether to enable this exporter | -| data-sources | strings | Yes | The range of data sources supported by different protocols. Values for ClickHouse `flow_metrics.*/flow_log.*/event.perf_event` data, also used for Kafka topic names | -| endpoints | strings | Yes | Remote receiving addresses, different protocol address formats vary, randomly select one that can be sent successfully | -| extra-headers | map | No | Header fields for remote HTTP requests, such as tokens for authentication can be added here | -| batch-size | int | No | Batch size, when this value is reached, send in batches. Default: 1024 (for protocol opentelemetry, default: 32) | -| flush-timeout | int | No | Flush interval, when this time is reached, send directly. Unit: seconds, default: 10 | -| queue-count | int | No | Concurrent send count, default: 4 | -| tag-filters | structs | No | Filter data that does not meet the conditions, only send data that meets the conditions. Default: empty, meaning no filtering. Detailed configuration below | -| export-fields | strings | Yes | Filter the fields or categories to be sent, only send fields that meet the conditions, such as: $tag, $metrics, meaning all fields in the `$tag` and `$metrics` categories are sent. Detailed configuration below | -| export-empty-tag | bool | No | Whether to send fields with empty tag values. Default: false, meaning not to send | -| export-empty-metrics-disabled | bool | No | Whether to send fields with metrics values of 0. Default: false, meaning to send | -| enum-translate-to-name-disabled | bool | No | Whether to translate enum type ID values to strings for sending. Default: false, meaning to translate | -| universal-tag-translate-to-name-disabled | bool | No | Whether to translate universal-tag type resource ID values to resource names for sending. Default: false, meaning to translate | -| sasl | struct | No | Used for Kafka protocol. Connection authentication method for Kafka, currently only supports 'SASL_SSL' with 'PLAIN' mechanism | -| topic | string | No | Used for Kafka protocol. Topic name, if empty, defaults to `deepflow.$data-source`, such as `deepflow.flow_log.l7_flow_log` | +| Field | Type | Required | Description | +| ---------------------------------------- | ------- | -------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | +| protocol | string | Yes | Supports `opentelemetry`, `prometheus`, `kafka` | +| enabled | bool | Yes | Whether to enable this exporter | +| data-sources | strings | Yes | Supported data source ranges vary by protocol. For ClickHouse: `flow_metrics.*/flow_log.*/event.perf_event` data, also used as Kafka topic names | +| endpoints | strings | Yes | Remote receiving addresses, format varies by protocol, randomly selects one that can send successfully | +| extra-headers | map | No | HTTP request headers for the remote endpoint, e.g., for authentication you can add tokens here | +| batch-size | int | No | Batch size; when this value is reached, data is sent in batches. Default: 1024 (for opentelemetry protocol, default: 32) | +| flush-timeout | int | No | Flush interval; when this time is reached, data is sent immediately. Unit: seconds, default: 10 | +| queue-count | int | No | Number of concurrent sends, default: 4 | +| tag-filters | structs | No | Filters out data that does not meet the conditions, only sends matching data. Default: empty, meaning no filtering. See detailed config below | +| export-fields | strings | Yes | Filters the fields or categories to send, only sends matching fields. For example: `$tag`, `$metrics` means all `$tag` and `$metrics` fields | +| export-empty-tag | bool | No | Whether to send fields with empty tag values. Default: false, meaning do not send | +| export-empty-metrics-disabled | bool | No | Whether to send fields with metrics value 0. Default: false, meaning send | +| enum-translate-to-name-disabled | bool | No | Whether to translate enum type ID values to strings before sending. Default: false, meaning translate | +| universal-tag-translate-to-name-disabled | bool | No | Whether to translate universal-tag type resource ID values to resource names before sending. Default: false, meaning translate | +| sasl | struct | No | For Kafka protocol. Kafka authentication method, currently only supports 'SASL_SSL' with 'PLAIN' mechanism | +| topic | string | No | For Kafka protocol. Topic name; if empty, defaults to `deepflow.$data-source`, e.g., `deepflow.flow_log.l7_flow_log` | ## tag-filters -Equivalent to the WHERE clause in SQL statements, see the yaml example below +Functions like the WHERE clause in SQL. Example YAML below: - `field-name` only supports original Tag field names in ClickHouse -- `field-values` do not support filling in string values of Resource type Tags -- `operator` includes +- `field-values` does not support filling in string values of Resource type Tags +- `operator` includes: - Numeric equal/not equal, string equal/not equal: `=`, `!=` - Numeric in/not in a set, string in/not in a set: `IN`, `NOT IN` - - String equal/not equal, supports * wildcard: `:`, `!:` - - String equal/not equal, supports regex: `~`, `!~` -- All tag-filters are logically AND + - String equal/not equal with `*` wildcard: `:`, `!:` + - String equal/not equal with regex: `~`, `!~` +- All tag-filters are combined with AND logic ```yaml tag-filters: @@ -108,12 +108,12 @@ Equivalent to the WHERE clause in SQL statements, see the yaml example below ## export-fields -Equivalent to the SELECT field-names clause in SQL statements, only supports using original field names in ClickHouse, see the yaml example below +Functions like the SELECT field-names clause in SQL, only supports using original field names in ClickHouse. Example YAML below: -- ``: Tag name, Metric name, such as ip4_0, ip4_1, region_id_0, region_id_1 -- ``: Field Category name, including `$tag`, `$k8s.label` and `$metrics` +- ``: Tag name or Metric name, e.g., ip4_0, ip4_1, region_id_0, region_id_1 +- ``: Field category name, including `$tag`, `$k8s.label`, and `$metrics` - `$tag`: All Tag fields - - `$tag.`: A specific category of Tag fields, `` as follows + - `$tag.`: A specific type of Tag fields, `` as follows - `flow_info` - \_id, time(s), start_time(us), end_time(us), close_type, flow_id, is_new_flow, status - `universal_tag` @@ -121,7 +121,7 @@ Equivalent to the SELECT field-names clause in SQL statements, only supports usi - l3_device_type[_0/1], l3_device_id[_0/1], l3_epc_id[_0/1], epc_id[_0/1], subnet_id[_0/1], service_id[_0/1] - auto_instance_id[_0/1], auto_instance_type[_0/1], auto_service_id[_0/1], auto_service_type[_0/1], gprocess_id[_0/1] - `native_tag` - - attribute_names, attribute_values + - attribute_names, attritube_values - `network_layer` - ip4[_0/1], ip6[_0/1], is_ipv4, protocol, province[_0/1] - `tunnel_info` @@ -145,7 +145,7 @@ Equivalent to the SELECT field-names clause in SQL statements, only supports usi - `data_link_layer` - mac[_0/1], eth_type, vlan - `$metrics`: All Metrics fields - - `$metrics.`: A specific category of Metrics fields, `` as follows + - `$metrics.`: A specific type of Metrics fields, `` as follows - `l3_throughput` - packet_tx, packet_rx, byte_tx, byte_rx, l3_byte_tx, l3_byte_rx, total_packet_tx, total_packet_rx, total_byte_tx, total_byte_rx - `l4_throughput` @@ -165,8 +165,8 @@ Equivalent to the SELECT field-names clause in SQL statements, only supports usi - `delay` - response_duration, duration, rtt, rtt_client, rtt_server, tls_rtt - rtt_max, rtt_client_max, rtt_server_max, srt_max, art_max, rrt_max, cit_max - - rtt_sum, rtt_client_sum, rtt_server_sum, srt_sum, art_sum, rrt_sum, cit_sum - - rtt_count, rtt_client_count, rtt_server_count, srt_count, art_count, rrt_count, cit_count + - rtt_sum, rtt_client_sum, rtt_server_sum, srt_sum, art_sum, rtt_sum, cit_sum + - rtt_count, rtt_client_count, rtt_server_count, srt_count, art_count, rtt_count, cit_count - `$k8s.label`: All k8s.label fields - `~$k8s.label.`: k8s.label sub-field names support regex diff --git a/translate/translated/09-diagnose/01-FAQ.md b/translate/translated/09-diagnose/01-FAQ.md index adada660..22d25365 100644 --- a/translate/translated/09-diagnose/01-FAQ.md +++ b/translate/translated/09-diagnose/01-FAQ.md @@ -7,68 +7,68 @@ permalink: /diagnose/FAQ # Deployment -1. What is the difference between all-in-one mode deployment and regular deployment? +1. What is the difference between all-in-one deployment mode and regular deployment? - Answer: All-in-one means that the storage components `clickhouse` and `mysql` do not have corresponding PVCs and are deployed using the `hostPath` mode. If the K8S cluster has multiple nodes, after restarting the `deepflow-clickhouse/mysql` Pods, they may drift to other nodes, causing previously collected data to be unqueryable. It is recommended to use all-in-one deployment for experience purposes, and regular deployment mode for testing/POC stages. + A: All-in-one means that the storage components `clickhouse` and `mysql` do not have corresponding PVCs and are deployed using the `hostPath` mode. If the K8S cluster has multiple nodes, after restarting the `deepflow-clickhouse/mysql` Pod, it may drift to another node, causing previously collected data to be unavailable for query. It is recommended to use all-in-one deployment for trial purposes, and use regular deployment mode for testing/POC phases. 2. How long is the data generally retained, and can it be adjusted? - Answer: The retention period for different data varies. You can check the retention period for different types of data in [server.yaml](https://github.com/deepflowio/deepflow/blob/main/server/server.yaml#L296-L310) and adjust the retention period `before the first deployment`. Modify the default configuration during [helm installation](../best-practice/server-advanced-config/#%E4%BF%AE%E6%94%B9-server-%E9%85%8D%E7%BD%AE%E6%96%87%E4%BB%B6) and complete the installation. + A: The retention period varies for different types of data. You can check the retention periods for different data types in [server.yaml](https://github.com/deepflowio/deepflow/blob/main/server/server.yaml#L296-L310), and adjust them `before the first deployment` by modifying the default configuration during [helm installation](../best-practice/server-advanced-config/#%E4%BF%AE%E6%94%B9-server-%E9%85%8D%E7%BD%AE%E6%96%87%E4%BB%B6). -3. How to use external MySQL/Clickhouse? +3. How to use external MySQL/ClickHouse? - Answer: Refer to the sections [Using Managed MySQL](../best-practice/production-deployment/#%E4%BD%BF%E7%94%A8%E6%89%98%E7%AE%A1-mysql) and [Using Managed ClickHouse](../best-practice/production-deployment/#%E4%BD%BF%E7%94%A8%E6%89%98%E7%AE%A1-clickhouse) in the [Production Deployment Recommendations](../best-practice/production-deployment/). + A: See the sections [Using Managed MySQL](../best-practice/production-deployment/#%E4%BD%BF%E7%94%A8%E6%89%98%E7%AE%A1-mysql) and [Using Managed ClickHouse](../best-practice/production-deployment/#%E4%BD%BF%E7%94%A8%E6%89%98%E7%AE%A1-clickhouse) in [Production Deployment Recommendations](../best-practice/production-deployment/). -4. The deployment specifications include two storage components, `mysql` and `clickhouse`. What is the difference between them? +4. The deployment specification includes two storage components, `mysql` and `clickhouse`. What is the difference between them? - Answer: `mysql` stores metadata information obtained from the deployment cluster, such as virtual machines, K8S resources, and synchronized collector information. `clickhouse` stores real-time collected data, such as network flow logs collected from the cluster, and performs aggregation analysis. + A: `mysql` stores metadata obtained from the deployed cluster, such as virtual machines, K8S resources, and synchronized collector information. `clickhouse` stores real-time collected data, such as network flow logs collected from the cluster, and performs aggregation and analysis. -5. After deployment, there is no data on Grafana? +5. No data on Grafana after deployment? - Answer: Please troubleshoot by following these steps: + A: Please troubleshoot as follows: - - Check if all Pods are running normally: Execute the `kubectl get pods -n deepflow` command and confirm that all Pods are in the `Running` state. + - Check whether all Pods are running normally: run `kubectl get pods -n deepflow` and confirm that all Pods are in the `Running` state. - - Check if DeepFlow Agent and DeepFlow Server are successfully connected. You can check if the service domain has been successfully created using the `deepflow-ctl domain list` command and check if the `STATE` is in the `NORMAL` state using the `deepflow-ctl agent list` command. + - Check whether DeepFlow Agent and DeepFlow Server are connected successfully: run `deepflow-ctl domain list` to check if a service domain has been successfully created, and run `deepflow-ctl agent list` to check whether the `STATE` is `NORMAL`. - - If there is no data in the `Network - X` type dashboard, check if the network card name matches the capture rules. You can view the default capture range using the `deepflow-ctl agent-group-config example | grep tap_interface_regex` command. If you are using a custom CNI or have set up the network in other ways, you can add the network card matching rules to `tap_interface_regex` and complete the modification by [updating the agent configuration](../best-practice/agent-advanced-config/#%E6%9B%B4%E6%96%B0-agent-group-config-%E9%85%8D%E7%BD%AE). + - If dashboards of the `Network - X` type have no data, check whether the NIC name matches the capture rules. You can run `deepflow-ctl agent-group-config example | grep tap_interface_regex` to view the default capture range. If you are using a custom CNI or have set up the network in another way, add the NIC matching rule to `tap_interface_regex` and [update the agent configuration](../best-practice/agent-advanced-config/#%E6%9B%B4%E6%96%B0-agent-group-config-%E9%85%8D%E7%BD%AE). - - If there is no data in the `Application - X` type dashboard, confirm that the application protocols used in the cluster meet the [supported list](../features/universal-map/request-log/). + - If only dashboards of the `Application - X` type have no data, confirm that the application protocols used in the cluster meet the [supported list](../features/universal-map/request-log/). -6. I have configured OpenTelemetry data integration/want to use DeepFlow's eBPF tracing and network tracing capabilities, but there is no data in the `Distributed Tracing` dashboard? +6. I have configured OpenTelemetry data integration / want to use DeepFlow's eBPF tracing and network tracing capabilities, but there is no data in the `Distributed Tracing` dashboard? - Answer: Please troubleshoot by following these steps: + A: Please troubleshoot as follows: - Using OpenTelemetry integration: - Confirm that the application has integrated the OTel SDK or started the OTel Agent. - - Confirm that the configuration has been completed according to the steps in [Configuring DeepFlow](../integration/input/tracing/opentelemetry/#%E9%85%8D%E7%BD%AE-deepflow). You can check if this feature is started normally on the container node where `deepflow-agent` is located using the `netstat -alntp | grep 38086` command. If the configuration is completed, you can check if there are flow logs with `Server Port` 38086 in `Network - Flow Log`. + - Confirm that you have completed the configuration according to [Configure DeepFlow](../integration/input/tracing/opentelemetry/#%E9%85%8D%E7%BD%AE-deepflow). On the container node where `deepflow-agent` is located, run `netstat -alntp | grep 38086` to check whether this feature has started normally. If configured, check in `Network - Flow Log` whether there are flow logs with `Server Port` 38086. - - Check if there is traffic from the application to the otel-agent to the container node in the `Application - K8s Pod Map` dashboard to ensure that this network link is smooth and requests are occurring. + - In the `Application - K8s Pod Map` dashboard, check whether there is traffic from the application to the otel-agent to the container node, ensuring that the network path is smooth and requests are occurring. - - Confirm in the `Application - Request Log` dashboard if there are any anomalies in the sent requests. + - In the `Application - Request Log` dashboard, confirm whether there are any anomalies in the requests sent. - Using eBPF capabilities: - Confirm that the server kernel version [meets the requirements](../ce-install/overview/#%E8%BF%90%E8%A1%8C%E6%9D%83%E9%99%90%E5%8F%8A%E5%86%85%E6%A0%B8%E8%A6%81%E6%B1%82). - - Check all replicas of `deepflow-agent`: Check if the eBPF module is started normally using the `kubectl logs -n deepflow ds/deepflow-agent | grep 'ebpf collector'` command, and confirm that the eBPF Tracer function is running normally using the `kubectl logs -n deepflow ds/deepflow-agent | grep TRACER` command. + - Check all replicas of `deepflow-agent`: run `kubectl logs -n deepflow ds/deepflow-agent | grep 'ebpf collector'` to check whether the eBPF module has started normally, and run `kubectl logs -n deepflow ds/deepflow-agent | grep TRACER` to confirm that the eBPF Tracer function is running normally. # Product -1. What should I do after installation and deployment? Are there any product cases or usage scenarios to share? +1. After installation and deployment, what should I do next? Are there any product cases or usage scenarios to share? - Answer: You can see the cases we share in our [Starting Observability](https://deepflow.io/blog/tags/Dashboard/) series of blogs and [troubleshooting](https://deepflow.io/blog/tags/troubleshooting/) series of blogs. You can also review past shares on our [Bilibili account](https://space.bilibili.com/2040480780/video). + A: You can find our shared cases in the [Getting Started with Observability](https://deepflow.io/blog/tags/Dashboard/) blog series and the [troubleshooting](https://deepflow.io/blog/tags/troubleshooting/) blog series. You can also review past sharing sessions on our [Bilibili account](https://space.bilibili.com/2040480780/video). 2. I think some features are not good enough and want to give suggestions. How can I do that? - Answer: You are welcome to submit a Feature Request on [Github Issue](https://github.com/deepflowio/deepflow/issues). If you already have a mature idea, you can also put it into practice directly and submit it in [GithubPR](https://github.com/deepflowio/deepflow/pulls). + A: You are welcome to submit a Feature Request in [Github Issue](https://github.com/deepflowio/deepflow/issues). If you already have a mature idea, you can also put it into practice directly and submit it in [Github PR](https://github.com/deepflowio/deepflow/pulls). -3. Where can I track the latest developments of DeepFlow? +3. Where can I follow the latest updates of DeepFlow? - Answer: You can check our latest release overview in the [Release Notes](../release-notes/release-6.2-ce/) or follow our latest [blogs](https://deepflow.io/blog/). + A: You can check our latest release overview in the [Release Notes](../release-notes/release-6.2-ce/) or follow our latest [blog](https://deepflow.io/blog/). # Contact Us -If the above help does not solve your problem, you can submit an issue via [Github Issue](https://github.com/deepflowio/deepflow/issues) or directly [contact us](https://github.com/deepflowio/deepflow#contact-us) for communication. \ No newline at end of file +If the above information does not solve your problem, you can submit an issue via [Github Issue](https://github.com/deepflowio/deepflow/issues) or [contact us directly](https://github.com/deepflowio/deepflow#contact-us) for communication. \ No newline at end of file diff --git a/translate/translated/09-diagnose/02-grafana.md b/translate/translated/09-diagnose/02-grafana.md index be2bdac4..15e7a0e8 100644 --- a/translate/translated/09-diagnose/02-grafana.md +++ b/translate/translated/09-diagnose/02-grafana.md @@ -7,14 +7,14 @@ permalink: /diagnose/grafana # Dashboard Panel Cannot Find Kubernetes Resources -> The specific manifestation is shown in the figure below: In the panels related to Pods, some Pods or namespaces are missing or lost (it may be one or two, or it could be many). +> The specific manifestation is as shown in the figure below: In panels related to Pods, some Pods or namespaces are missing or lost (it could be one or two, or it could be many). **Step 1. Check the status of deepflow-agent in the cluster** ``` - ## Check the running status of the agent on the deepflow server side, NORMAL indicates normal operation + ## On the deepflow server side, check the running status of the agent; NORMAL means normal deepflow-ctl agent list ``` @@ -23,7 +23,7 @@ permalink: /diagnose/grafana ``` ## Add environment variables to the agent pod: - ## After running, grep replicasets in the logs, and check whether the total time displayed for querying each page (kubernetes-api-list-limit) in DEBUG logs is > 5min + ## After running, grep replicasets in the logs, and check in the DEBUG log whether the total time taken to query each page (kubernetes-api-list-limit) exceeds 5 minutes - name: RUST_LOG value=info,deepflow_agent::platform::kubernetes::resource_watcher=debug ``` @@ -31,17 +31,17 @@ permalink: /diagnose/grafana If the log output is too much and inconvenient to view, you can directly check the synchronized data through the DeepFlow System - DeepFlow Agent panel. -**Step 3. Solution for excessive resource synchronization** +**Step 3. Solution for excessive synchronized resources** -> As mentioned in Step 2, deepflow-agent synchronizes 1000 k8s resource information by default each time, while the default expiration time for the k8s continue token is 5 minutes. Exceeding this time will cause synchronization interruption. -> The role of the continue token: +> As described in Step 2, deepflow-agent synchronizes 1000 pieces of k8s resource information by default each time, while the default expiration time for the k8s continue token is 5 minutes. Exceeding this time will cause synchronization to be interrupted. +> The purpose of the continue token: > > - https://kubernetes.io/zh-cn/docs/reference/kubernetes-api/common-parameters/common-parameters/#continue > - https://kubernetes.io/zh-cn/docs/reference/using-api/api-concepts/#retrieving-large-results-sets-in-chunks -Solutions: +Solution: -- Solution 1: Increase the single-page query limit (kubernetes-api-list-limit) +- Option 1: Increase the number of items queried per page (kubernetes-api-list-limit) https://github.com/deepflowio/deepflow/blob/main/server/agent_config/example.yaml#L468 -- Solution 2: Increase the continue token expiration time (--etcd-compaction-interval) - https://stackoverflow.com/questions/63664353/how-to-modify-default-expired-time-of-continue-token-in-kubernetes +- Option 2: Increase the continue token expiration time (--etcd-compaction-interval) + https://stackoverflow.com/questions/63664353/how-to-modify-default-expired-time-of-continue-token-in-kubernetes \ No newline at end of file diff --git a/translate/translated/09-diagnose/03-deepflow-agent.md b/translate/translated/09-diagnose/03-deepflow-agent.md index 8e79e718..0f0dfec2 100644 --- a/translate/translated/09-diagnose/03-deepflow-agent.md +++ b/translate/translated/09-diagnose/03-deepflow-agent.md @@ -5,17 +5,17 @@ permalink: /diagnose/agent > This document was translated by ChatGPT -# Agent Error Due to Significant System Time Rollback +# Significant System Time Rollback Causes Agent Errors ## Symptoms -Panic appears in the agent logs with the term `SystemTimeError`. +A panic appears in the agent logs, containing the term `SystemTimeError`. ## Cause -When the agent calculates the time difference, if the subtracted time is older (resulting in a negative time difference), the unprotected code will report an error. +When the agent calculates the time difference, if the subtracted time is earlier (resulting in a negative time difference), unprotected code will throw an error. ## Solution -- Use ntpd or similar methods to slowly adjust the system time. -- If making a significant time adjustment, the agent needs to be restarted. \ No newline at end of file +- Use tools like ntpd to gradually adjust the system time. +- If the time is adjusted significantly, restart the agent. \ No newline at end of file diff --git a/translate/translated/09-diagnose/04-deepflow-server.md b/translate/translated/09-diagnose/04-deepflow-server.md index 364db8b5..842c1f8f 100644 --- a/translate/translated/09-diagnose/04-deepflow-server.md +++ b/translate/translated/09-diagnose/04-deepflow-server.md @@ -5,97 +5,97 @@ permalink: /diagnose/how-to-profile-deepflow-server-ingester > This document was translated by ChatGPT -# Investigating Packet Loss in deepflow-server +# Investigating Packet Loss Causes in deepflow-server -- Influencing Factors: +- Influencing factors: - - Server Performance Bottleneck - - Check CPU and memory bottlenecks through the DeepFlow Server Dashboard. If they are maxed out, it indicates a non-server bottleneck. - - Clickhouse Performance Bottleneck + - Server performance bottleneck + - Check CPU and memory bottlenecks via the DeepFlow Server Dashboard. If they are fully utilized, it indicates a non-server bottleneck. + - ClickHouse performance bottleneck - - Use the following query to determine Clickhouse write performance: + - Use the following queries to determine ClickHouse write performance: ```SQL - -- Single Clickhouse reception performance: - -- written_rows / (query_duration_ms/1000) * server write thread count (ingester.flow-ck-writer.queue_count) + -- Receiving performance of a single ClickHouse: + -- written_rows / (qurey_duration_ms/1000) * server write thread count (ingester.flow-ck-writer.queue_count) - SELECT event_time, query_duration_ms, written_rows, written_bytes, query + SELECT event_time,query_duration_ms,written_rows,written_bytes,query FROM system.query_log - WHERE event_time > now() - 1000 AND query LIKE '%INSERT INTO%' + WHERE event_time>now()-1000 AND query LIKE '%INSERT INTO%' ORDER BY query_duration_ms - DESC LIMIT 10 + DESC limit 10 ``` ```SQL - -- Average server write to Clickhouse per second + -- Average data written per second from server to ClickHouse SELECT tag_values[1] AS host, AVG(metrics_float_values[4])/10 AS written_per_s, AVG(metrics_float_values[3])/10 AS drop_per_s FROM deepflow_system.deepflow_system - WHERE virtual_table_name = 'deepflow_server_ingester_ckwriter' + WHERE virtual_table_name='deepflow_server_ingester_ckwriter' GROUP BY host ORDER BY written_per_s - DESC LIMIT 30; + DESC limit 30; ``` ```SQL - -- Server packet loss queue view + -- How to check server packet loss queues SELECT tag_values[1] AS host, tag_values[3] AS queue, AVG(metrics_float_values[1])/10 AS avg_total_per_s, AVG(metrics_float_values[2])/10 AS avg_handled_per_s, AVG(metrics_float_values[3])/10 AS avg_drop_per_s FROM deepflow_system.deepflow_system - WHERE virtual_table_name = 'deepflow_server_ingester_queue' AND time > now() - 900 - GROUP BY host, queue - ORDER BY avg_drop_per_s, avg_total_per_s - DESC LIMIT 30; + WHERE virtual_table_name='deepflow_server_ingester_queue' AND time > now()-900 + GROUP BY host,queue + ORDER BY avg_drop_per_s,avg_total_per_s + DESC limit 30; ``` -- Check packet loss when the server writes to Clickhouse: +- View packet loss when the server writes to ClickHouse: - | Queue Name | Queue Count Configuration | Queue Length Configuration | Queue Description | - | --- | ---- | --- | ---- | - | 1-recv-unmarshall | unmarshall-queue-count | unmarshall-queue-size | Metric data processing queue | - | 1-receive-to-decode-l4/l7 | flow-log-decoder-queue-count | flow-log-decoder-queue-size | Flow log processing queue | - | 1-receive-to-decode-telegraf/prometheus/deepflow_stats | ext-metrics-decoder-queue-count | ext-metrics-decoder-queue-size | Other data processing queue | - | 1-receive-to-decode-profile | profile-decoder-queue-count | profile-decoder-queue-size | Performance analysis data processing queue | - | 1-receive-to-decode-proc_event | perf-event-decoder-queue-count | perf-event-decoder-queue-size | IO and other event processing queue | - | 1-receive-to-decode-raw_pcap | pcap-queue-count | pcap-queue-size | pcap packet processing queue | - | flow_metrics- prefix | metrics-ck-writer->queue-count | metrics-ck-writer->queue-size | Metric data write queue | - | flow_log-l7_packet prefix | pcap-ck-writer->queue-count | pcap-ck-writer->queue-size | pcap data write queue | - | flow_log- prefix except flow_log-l7_packet | flowlog-ck-writer->queue-count | flowlog-ck-writer->queue-size | Flow log data write queue | - | ext_metrics- prefix | ext_metrics-ck-writer->queue-count | ext_metrics-ck-writer->queue-size | Other data write queue | - | profile- prefix | profile-ck-writer->queue-count | profile-ck-writer->queue-size | Performance analysis data write queue | + | Queue Name | Queue Count Config | Queue Length Config | Queue Description | + | --- | ---- | --- | ---- | + | 1-recv-unmarshall | unmarshall-queue-count | unmarshall-queue-size | Metrics data processing queue | + | 1-receive-to-decode-l4/l7 | flow-log-decoder-queue-count | flow-log-decoder-queue-size | Flow log processing queue | + | 1-receive-to-decode-telegraf/prometheus/deepflow_stats | ext-metrics-decoder-queue-count | ext-metrics-decoder-queue-size | Other data processing queue | + | 1-receive-to-decode-profile | profile-decoder-queue-count | profile-decoder-queue-size | Performance analysis data processing queue | + | 1-receive-to-decode-proc_event | perf-event-decoder-queue-count | perf-event-decoder-queue-size | IO and other event processing queue | + | 1-receive-to-decode-raw_pcap | pcap-queue-count | pcap-queue-size | pcap packet processing queue | + | flow_metrics- prefix | metrics-ck-writer->queue-count | metrics-ck-writer->queue-size | Metrics data write queue | + | flow_log-l7_packet prefix | pcap-ck-writer->queue-count | pcap-ck-writer->queue-size | pcap data write queue | + | flow_log- prefix except flow_log-l7_packet | flowlog-ck-writer->queue-count | flowlog-ck-writer->queue-size | Flow log data write queue | + | ext_metrics- prefix | ext_metrics-ck-writer->queue-count | ext_metrics-ck-writer->queue-size| Other data write queue | + | profile- prefix | profile-ck-writer->queue-count | profile-ck-writer->queue-size | Performance analysis data write queue | - Handling other types of packet loss: - | Type | Metric Set | Metric | - | ------------------- | ------------------ | ---------------------------- | - | Queue Packet Loss | ingester.queue | metrics.overwritten | - | Flow Log Sampling Loss | ingester.decoder | metrics.drop_count | - | Data Write Packet Loss | ingester.ckwriter | metrics.write_failed_count | - | Invalid Data Packet Loss | ingester.receiver | metrics.invalid | - - - Handling Flow Log Sampling Loss: - - Check the corresponding queue packet loss through the Dashboard: DeepFlow Server - Ingester in the flow log (throttle-drop) panel. - - By default, L4/L7 flow log processing is 50k/s. If CPU, memory, and disk are sufficient, you can increase the throttle to enhance processing capacity. - - Adjust the configuration parameters of the [Ingester](https://github.com/deepflowio/deepflow/blob/main/server/server.yaml#L347) module to increase processing capacity and avoid packet loss. - - Handling Data Write Packet Loss: - - Filter server logs for `write block failed` to see the reason for write failures. - - If using PV, check if there is available space in the backend storage. - - If using hostPath, check if there is available space on the local disk. - - Handling Invalid Data Packet Loss: - - Filter server logs for `TCP client` to get the IP address of the invalid data sender. - - If sent by DeepFlow-Agent, confirm whether the DeepFlow-Agent and DeepFlow-Server versions are consistent. - - If not sent by DeepFlow-Agent, block the IP from sending data to the data node's listening port (default: 30033), or increase the alert threshold to suppress alerts generated by such IPs sending data. + | Type | Metric Set | Metric | + | --------------------- | ------------------ | ---------------------------- | + | Queue packet loss | ingester.queue | metrics.overwritten | + | Flow log sampling loss| ingester.decoder | metrics.drop_count | + | Data write loss | ingester.ckwriter | metrics.write_failed_count | + | Invalid data loss | ingester.recviver | metrics.invalid | + + - Handling flow log sampling loss: + - View the corresponding queue packet loss in the `flow log(throttle-drop)` Panel in Dashboard: DeepFlow Server - Ingester. + - By default, L4/L7 flow log processing is 50,000/s. If CPU, memory, and disk are sufficient, you can increase the throttle to improve processing capacity. + - Adjust the configuration parameters of the [Ingester](https://github.com/deepflowio/deepflow/blob/main/server/server.yaml#L347) module on the data node to increase processing capacity and avoid packet loss. + - Handling data write loss: + - Filter server logs for `write block failed` to check the reason for write failures. + - If using PV, check whether the backend storage still has available space. + - If using hostPath, check whether the local disk still has available space. + - Handling invalid data loss: + - Filter server logs for `TCP client` to obtain the IP address of the invalid data sender. + - If sent by DeepFlow-Agent, verify that the DeepFlow-Agent and DeepFlow-Server versions match. + - If not sent by DeepFlow-Agent, block that IP from sending data to the data node's listening port (default: 30033), or raise the alert threshold to suppress alerts caused by such IPs sending data. # Introduction -Using [Golang Profile](https://go.dev/blog/pprof), we can capture and analyze the data write performance of DeepFlow Server for optimization. +Through [Golang Profile](https://go.dev/blog/pprof), we can capture and analyze DeepFlow Server's data write performance for optimization. # Steps 1. Install the [deepflow-ctl](../ce-install/upgrade/#%E5%8D%87%E7%BA%A7-deepflow-cli) tool. -2. Find the DeepFlow Server Pod IP that needs Profile analysis. If the number of DeepFlow Server replicas is greater than 1, select any one of them: +2. Find the Pod IP of the DeepFlow Server to be analyzed with Profile. If the number of DeepFlow Server replicas is greater than 1, select any one of them: ```bash deepflow_server_pod_ip=$(kubectl -n deepflow get pods -o wide | grep deepflow-server | awk '{print $6}') @@ -113,7 +113,7 @@ deepflow-ctl -i $deepflow_server_pod_ip ingester profiler on go tool pprof http://$deepflow_server_pod_ip:9526/debug/pprof/profile ``` -After executing the command, the default sampling time is 30s. You can modify the Profile duration by adding the `seconds=x` parameter, such as `http://$deepflow_server_pod_ip:9526/debug/pprof/profile?seconds=60`. After the Profile ends, you can enter the `svg` command to generate a vector format Profile result graph and copy it locally to view it through a browser. +After executing the command, the default sampling time is 30s. You can modify the Profile duration by adding the `seconds=x` parameter, e.g., `http://$deepflow_server_pod_ip:9526/debug/pprof/profile?seconds=60`. After the Profile ends, you can enter the `svg` command to generate a vector graphic of the Profile result and copy it locally to view in a browser. # Get Memory Profile @@ -121,8 +121,8 @@ After executing the command, the default sampling time is 30s. You can modify th go tool pprof http://$deepflow_server_pod_ip:9526/debug/pprof/heap ``` -After executing the command, real-time sampling will be performed to obtain the current memory snapshot. Similarly, you can enter the `svg` command to generate a vector format Profile result graph and copy it locally to view it through a browser. +After executing the command, real-time sampling will be performed to obtain the current memory snapshot. Similarly, you can enter the `svg` command to generate a vector graphic of the Profile result and copy it locally to view in a browser. # Other Profile Information -If you want to obtain other Profile information, you can find all types available for analysis in the [Golang SourceCode](https://github.com/golang/go/blob/master/src/net/http/pprof/pprof.go#L350). \ No newline at end of file +If you want to obtain other Profile information, you can find all available analysis types in the [Golang SourceCode](https://github.com/golang/go/blob/master/src/net/http/pprof/pprof.go#L350). \ No newline at end of file diff --git a/translate/translated/10-release-notes/01-versioning.md b/translate/translated/10-release-notes/01-versioning.md index 15d70a3f..9babe1c9 100644 --- a/translate/translated/10-release-notes/01-versioning.md +++ b/translate/translated/10-release-notes/01-versioning.md @@ -7,15 +7,15 @@ permalink: /release-notes/versioning # Version Naming -DeepFlow adheres to the [Semantic Versioning](https://semver.org/) method for version naming. The version number format is `X.Y.Z`, where `X` stands for Major Version, `Y` represents Minor Version, and `Z` is the Patch Version. +DeepFlow follows the [Semantic Versioning](https://semver.org/) naming convention, with the version number format `X.Y.Z`, where `X` is the Major Version, `Y` is the Minor Version, and `Z` is the Patch Version. # Iteration Cycle -- The major version number X changes approximately every `two years` -- The minor version number Y changes approximately every `four months` -- The patch version number Z changes approximately every `two weeks` +- The major version number `X` changes approximately every **two years**. +- The minor version number `Y` changes approximately every **four months**. +- The patch version number `Z` changes approximately every **two weeks**. -# Version Maintenance Time +# Version Maintenance Period -- The largest `Z` version in each `X.Y` version is the long-term support (LTS, Long-Term Support) version -- After each `X.Y.Z` version is released, all `X.Y.0` - `X.Y.{Z-1}` versions are no longer maintained or updated +- The largest `Z` version within each `X.Y` release is designated as a Long-Term Support (LTS) version. +- After the release of each `X.Y.Z` version, all versions from `X.Y.0` to `X.Y.{Z-1}` will no longer be maintained or updated. \ No newline at end of file diff --git a/translate/translated/10-release-notes/02-release-timeline.md b/translate/translated/10-release-notes/02-release-timeline.md index 089e99e6..e2495b2d 100644 --- a/translate/translated/10-release-notes/02-release-timeline.md +++ b/translate/translated/10-release-notes/02-release-timeline.md @@ -7,16 +7,39 @@ permalink: /release-notes/release-timeline | Branch | Tag | LTS | Release | End Of Life | | ------ | ------- | ----- | -------------- | ------------------ | -| main | 6.6.8 | | 2024-09-19 (E) | | -| | 6.6.7 | | 2024-09-12 (E) | | -| | 6.6.6 | | 2024-09-05 (E) | | -| | 6.6.5 | | 2024-08-29 (E) | | -| | 6.6.4 | | 2024-08-22 (E) | | +| main | | Y | 2025-11-18 (E) | | +| | 7.1.7 | | 2025-11-11 (E) | | +| | 7.1.6 | | 2025-11-04 (E) | | +| | 7.1.5 | | 2025-10-28 (E) | | +| | 7.1.4 | | 2025-10-21 (E) | | +| | 7.1.3 | | 2025-10-14 (E) | | +| | 7.1.2 | | 2025-09-18 (E) | | +| | 7.1.1 | | 2025-09-04 (E) | | +| | 7.1.0 | | 2025-08-21 | | +| v7.0 | | Y | 2025-06-19 | 2026-06-19 | +| | 7.0.9 | | 2025-06-11 | | +| | 7.0.8 | | 2025-05-16 | | +| | 7.0.7 | | 2025-04-29 | | +| | 7.0.6 | | 2025-04-15 | | +| | 7.0.5 | | 2025-04-02 | | +| | 7.0.4 | | 2025-03-18 | | +| | 7.0.3 | | 2025-03-05 | | +| | 7.0.2 | | 2025-02-11 | | +| | 7.0.1 | | 2025-01-16 | | +| | 7.0.0 | | 2025-01-02 | | +| v6.6 | | Y | 2024-12-12 | 2026-12-12 | +| | 6.6.9 | | 2024-12-12 | | +| | 6.6.8 | | 2024-11-14 | | +| | 6.6.7 | | 2024-10-31 | | +| | 6.6.6 | | 2024-10-11 | | +| | 6.6.5 | | 2024-09-24 | | +| | 6.6.4 | | 2024-08-29 | | | | 6.6.3 | | 2024-08-15 | | | | 6.6.2 | | 2024-08-01 | | | | 6.6.1 | | 2024-07-18 | | | | 6.6.0 | | 2024-07-04 | | -| v6.5 | 6.5.9 | Y | 2024-06-20 | 2025-06-20 | +| v6.5 | | Y | 2024-06-20 | 2025-06-20 | +| | 6.5.9 | | 2024-06-20 | | | | 6.5.8 | | 2024-06-06 | | | | 6.5.7 | | 2024-05-23 | | | | 6.5.6 | | 2024-05-10 | | @@ -26,7 +49,8 @@ permalink: /release-notes/release-timeline | | 6.5.2 | | 2024-03-12 | | | | 6.5.1 | | 2024-02-27 | | | | 6.5.0 | | 2024-02-06 | | -| v6.4 | 6.4.9 | Y | 2024-01-18 | 2025-01-25 | +| v6.4 | | ~~Y~~ | 2024-01-18 | 2025-01-25 | +| | 6.4.9 | | 2024-01-18 | | | | 6.4.8 | | 2024-01-11 | | | | 6.4.7 | | 2024-01-04 | | | | 6.4.6 | | 2023-12-28 | | @@ -36,7 +60,8 @@ permalink: /release-notes/release-timeline | | 6.4.2 | | 2023-11-09 | | | | 6.4.1 | | 2023-10-26 | | | | 6.4.0 | | 2023-10-12 | | -| v6.3 | 6.3.9 | Y | 2023-09-14 | 2024-09-21 | +| v6.3 | | ~~Y~~ | 2023-09-14 | 2024-09-21 | +| | 6.3.9 | | 2023-09-14 | | | | 6.3.8 | | 2023-09-07 | | | | 6.3.7 | | 2023-08-31 | | | | 6.3.6 | | 2023-08-24 | | @@ -46,7 +71,8 @@ permalink: /release-notes/release-timeline | | 6.3.2 | | 2023-06-29 | | | | 6.3.1 | | 2023-06-15 | | | | 6.3.0 | | 2023-06-01 | | -| v6.2 | 6.2.6.5 | Y | 2023-05-17 | 2024-05-24 | +| v6.2 | | ~~Y~~ | 2023-05-17 | 2024-05-24 | +| | 6.2.6.5 | | 2023-05-17 | | | | 6.2.6.4 | | 2023-05-10 | | | | 6.2.6.3 | | 2023-04-27 | | | | 6.2.6.2 | | 2023-04-20 | | diff --git a/translate/translated/10-release-notes/03-ce-6.6-release.md b/translate/translated/10-release-notes/03-ce-6.6-release.md deleted file mode 100644 index cd2f9432..00000000 --- a/translate/translated/10-release-notes/03-ce-6.6-release.md +++ /dev/null @@ -1,83 +0,0 @@ ---- -title: v6.6 CE Release Notes -permalink: /release-notes/release-6.6-ce ---- - -> This document was translated by ChatGPT - -# v6.6.3 [2024/08/15] - -## Beta Feature - -- AutoTracing - - When the TraceID is present in the protocol header, support disabling eBPF syscall_trace_id calculation (via configuration `syscall_trace_id_disabled`) to reduce the impact on business performance. - - Automatically correct minor clock deviations between different machines in the distributed tracing flame graph. -- AutoTagging - - Support using Lua Plugin to customize K8s workload abstraction rules, [documentation](../features/auto-tagging/k8s-crd/). -- Agent - - Support completely disabling cBPF data collection (via configuration `tap_interface_regex` as an empty string) to reduce memory overhead. - - Support deepflow-agent to use a single socket to transmit all observability data, and this feature can be disabled via the `multiple_sockets_to_ingester` configuration item to use multiple sockets to improve transmission performance. - -## Stable Feature - -- AutoProfiling - - Support viewing DeepFlow eBPF On-CPU Profiling data in the Grafana Panel, [Demo](https://ce-demo.deepflow.yunshan.net/d/Continuous_Profiling/continuous-profiling?var-app_service=deepflow-server). -- AutoMetrics - - Support aligning timestamps of request and response metrics in the same session to help AIOps systems better achieve root cause localization (thanks to `pegasusljn`:[FR](https://github.com/deepflowio/deepflow/issues/7069)). -- AutoTagging - - Correctly mark the Universal Tag for loopback network card traffic on K8s Node. -- Agent - - Reduce the number of sockets used by deepflow-agent to send data. - - Merge the sockets used to transmit open_telemetry and open_telemetry_compressed data when integrating OpenTelemetry. - - Merge the sockets used to transmit deepflow_stats and agent_log data for agent self-monitoring. - - Merge the sockets used to transmit prometheus and telegraf metrics when integrating Prometheus and Telegraf. - -# v6.6.2 [2024/08/01] - -## Beta Feature - -- AutoMetrics - - Support aligning timestamps of request and response metrics in the same session to help AIOps systems better achieve root cause localization (thanks to `pegasusljn`:[FR](https://github.com/deepflowio/deepflow/issues/7069)). - -## Stable Feature - -- AutoTracing - - Optimize the default values of configuration parameters for NTP clock offset (`host_clock_offset_us`) and network transmission delay (`network_delay_us`) used in network Span tracing to reduce the probability of mismatches. - -# v6.6.1 [2024/07/18] - -## Beta Feature - -- AutoTagging - - Correctly mark the Universal Tag for loopback network card traffic on K8s Node. - -## Stable Feature - -- AutoTracing - - Added URL desensitization capability for HTTP protocol, Redis protocol desensitization enabled by default. -- AutoTagging - - Support synchronizing Volcengine resource tags, [documentation](../features/auto-tagging/meta-tags/). - - Cancel synchronization of Pods in K8s Evicted state to reduce resource overhead. -- Integration - - Optimize the mapping of fields such as schema/target in OTel Span to `l7_flow_log`, [documentation](../features/l7-protocols/otel/). -- Agent - - Support aggregating and collecting traffic from multiple member physical network cards of the Open vSwitch bond interface. - -# v6.6.0 [2024/07/04] - -## Backward Incompatible Change - -- AutoProfiling - - Use Dataframe return format to compress response size and improve API performance, [PR](https://github.com/deepflowio/deepflow/pull/7011), [documentation](../features/continuous-profiling/data/). - -| | #Functions | Response Size (Byte) | Download Time | -| ------ | ---------- | -------------------- | ------------- | -| Before | 450,000 | 21.9M | 6.16s | -| After | 450,000 | 3.07M | 0.78s | - -## Beta Feature - -- AutoTagging - - Support synchronizing Volcengine resource tags, [documentation](../features/auto-tagging/meta-tags/). -- Agent - - Support aggregating and collecting traffic from multiple member physical network cards of the Open vSwitch bond interface. \ No newline at end of file diff --git a/translate/translated/10-release-notes/03-ce-7.1-release.md b/translate/translated/10-release-notes/03-ce-7.1-release.md new file mode 100644 index 00000000..748a827e --- /dev/null +++ b/translate/translated/10-release-notes/03-ce-7.1-release.md @@ -0,0 +1,24 @@ +--- +title: v7.1 CE Release Notes +permalink: /release-notes/release-7.1-ce +--- + +> This document was translated by ChatGPT + +# v7.1.0 [2025/08/21] + +## Stable Feature + +- AutoTracing + - Supports collecting multiple HTTP2/gRPC requests and responses within a single Packet. + - Supports retrieving the full file path for file read/write events: + - Fully retrieves NAS file paths, supporting protocols such as NFS, SMB, and CIFS. + - Fully retrieves the complete absolute path for file read/write operations inside container Pods. +- AutoMetrics + - Supports aggregating and generating eBPF profiling metric data at 1-second granularity, accelerating profiling metric queries. +- AutoTagging + - Simplified process synchronization blacklist configuration, [documentation](../configuration/agent/#inputs.proc.process_blacklist). + - Adapted to K8s v1.32+ API. +- Agent + - Supports the Watchdog mechanism to ensure proper execution of circuit breaking in extreme cases. + - Supports compressed transmission of application logs, with a compression ratio between 5:1 and 20:1, [documentation](../configuration/agent/#outputs.compression.application_log). \ No newline at end of file diff --git a/translate/translated/10-release-notes/04-ee-6.6-release.md b/translate/translated/10-release-notes/04-ee-6.6-release.md deleted file mode 100644 index 8d0759b3..00000000 --- a/translate/translated/10-release-notes/04-ee-6.6-release.md +++ /dev/null @@ -1,8 +0,0 @@ ---- -title: v6.6 EE Release Notes -permalink: /release-notes/release-6.6-ee ---- - -> This document was translated by ChatGPT - -TBD diff --git a/translate/translated/10-release-notes/04-ee-7.1-release.md b/translate/translated/10-release-notes/04-ee-7.1-release.md new file mode 100644 index 00000000..e41ee91f --- /dev/null +++ b/translate/translated/10-release-notes/04-ee-7.1-release.md @@ -0,0 +1,68 @@ +--- +title: v7.1 EE Release Notes +permalink: /release-notes/release-7.1-ee +--- + +> This document was translated by ChatGPT + +# Business and Applications + +## Business Observability + +N/A + +## Application Observability + +- AutoTracing + - Supports collecting multiple HTTP2/gRPC requests and responses within a single Packet. + - Supports retrieving the full file path for file read/write events: + - Fully retrieves NAS file paths, supporting NFS, SMB, CIFS, and other protocols. + - Fully retrieves the complete absolute path for file read/write operations inside container Pods. + +## Code Observability + +- AutoMetrics + - Supports aggregating and generating eBPF profiling metric data at 1-second granularity, accelerating profiling metric queries. + +# Infrastructure + +## Asset Observability + +N/A + +## Network Observability + +N/A + +## Traffic Distribution + +N/A + +# Customization + +## Dashboards + +N/A + +# Others + +## Usability + +- Supports copying and pasting the entire content of the search box. + +## Alert Management + +- Supports customizing the criteria for determining alert recovery: if there are no "critical/error/warning/no data" statuses in N consecutive monitoring events, a recovery event will be generated. + +## Resource List + +- AutoTagging + - Supports entering custom `cloud.tag` labels on the page. + - Simplified process synchronization blacklist configuration, [documentation](../configuration/agent/#inputs.proc.process_blacklist). + - Adapted to K8s v1.32+ API. + +## System Management + +- Agent + - Supports Watchdog mechanism to ensure circuit breaking can be executed properly under extreme conditions. + - Supports compressing application logs before sending, with a compression ratio between 5:1 and 20:1, [documentation](../configuration/agent/#outputs.compression.application_log). \ No newline at end of file diff --git a/translate/translated/10-release-notes/05-ce-7.0-release.md b/translate/translated/10-release-notes/05-ce-7.0-release.md new file mode 100644 index 00000000..dc7e266e --- /dev/null +++ b/translate/translated/10-release-notes/05-ce-7.0-release.md @@ -0,0 +1,120 @@ +--- +title: v7.0 CE Release Notes +permalink: /release-notes/release-7.0-ce +--- + +> This document was translated by ChatGPT + +# v7.0.9 [2025/06/11] + +## Stable Feature + +- AutoTracing + - Support collecting Unix Socket call logs (`l7_flow_log`), and enable automatic tracing between TCP/UDP Socket call logs and Unix Socket call logs. + - Support parsing SRV type DNS call logs, [Documentation](https://en.wikipedia.org/wiki/SRV_record). + - Optimize parsing of unary type gRPC calls, [Documentation](../configuration/agent/#processors.request_log.application_protocol_inference.protocol_special_config.grpc.streaming_data_enabled). +- AutoTagging + - Support collecting change events of K8s resource definitions and ConfigMaps. +- Server + - Support MCP Server. +- Agent + - When the Agent reaches the traffic rate limit, support choosing between `drop` or `wait` strategies. The default behavior is drop, but it can be configured to wait to improve data sending success rate, [Documentation](../configuration/agent/#global.communication.ingester_traffic_overflow_action). + - Added a circuit breaker mechanism for free disk space in the Agent runtime environment, [Documentation](../configuration/agent/#global.circuit_breakers.free_disk). + - Support disabling Agent's use of Swap memory, [Documentation](../configuration/agent/#global.tunning.swap_disabled). + - Adapt to K8s CNI with identical virtual NIC MAC addresses on the same host. + - Optimization: Reduce the work done by the Agent when it is disabled. + +# v7.0.8 [2025/05/16] + +## Stable Feature + +- AutoTracing + - Support parsing truncated MySQL protocol content. +- Agent + - Support compressing and sending call logs and flow logs. In test environments, call log compression ratio can reach 8:1, [Documentation](../configuration/agent/#outputs.compression.l7_flow_log). + +# v7.0.7 [2025/04/28] + +## Stable Feature + +- AutoTracing + - Enrich eBPF hook points for collecting file read/write events (`io_event`) to improve adaptability. +- AutoTagging + - Optimize the meaning of the `process_kname` field in call logs and file read/write event data, changing from `kernel thread name` to `system process` name for better readability. + - Aggregate processes with the same `cmdline` within the same cloud host or the same K8s workload into a unique gprocess to reduce redundant process information. + - Optimize default values of the process matcher, [Documentation](../configuration/agent/#inputs.proc.process_matcher). + - By default, ignore collection of process information for `sleep/sh/bash/pause/runc`. + - By default, collect process information and OnCPU profiling data for `Java/Python`, and automatically record the gprocess name as the jar/py file name to avoid all being displayed as java/python. + - By default, collect process information and OnCPU profiling data for `deepflow-*`. + - By default, collect process information inside containers. + - Optimize the meaning of the `response_status` field in call logs (`l7_flow_log`). + - **Normal**: Response code is normal. + - **Client Error**: Response code indicates a client-side error, e.g., HTTP 4XX. + - **Server Error**: Response code indicates a server-side error, e.g., HTTP 5XX. + - **Timeout**: If no response is collected within a certain time, the request will be marked as timed out. + - Agent `Application session merge timeout` configuration: DNS and TLS default 15s, other protocols default 120s, [Documentation](../configuration/agent/#processors.request_log.timeouts.session_aggregate). + - **Unknown**: When concurrent requests exceed the collector's cache capacity, the oldest requests will be marked as unknown. + - Agent `Maximum session aggregation entries` configuration: Default cache of 64K requests, [Documentation](../configuration/agent/#processors.request_log.tunning.session_aggregate_max_entries). + - **Parse Failed**: Response was collected but could not parse the response code due to truncation or compression. + - Agent `Payload truncation` configuration: By default, parse the first 1024 bytes of the Payload, [Documentation](../configuration/agent/#processors.request_log.tunning.payload_truncation). +- Agent + - Optimize memory usage of the cache for application performance metrics in the Agent by promptly clearing expired LRU entries, reducing overall memory consumption by **43%** in test environments. + - Aggregate and store flow logs (`l4_flow_log`) generated by LB health checks, reducing flow log storage overhead by nearly **50%** in some production environments, [Documentation](../configuration/agent/#outputs.flow_log.aggregators.aggregate_health_check_l4_flow_log). + - Optimize the resource overhead protection mechanism when application protocol recognition fails to avoid mistakenly disabling application protocol parsing, [Documentation](../configuration/agent/#processors.request_log.application_protocol_inference.inference_max_retries). + +# v7.0.6 [2025/04/15] + +## Stable Feature + +- AutoTracing + - Support parsing compressed MySQL calls, [Documentation](../configuration/agent/#processors.request_log.application_protocol_inference.protocol_special_config.mysql.decompress_payload). + - Support parsing MySQL Login Response statements. + - Support parsing multiple DNS requests in TCP Payload. +- Agent + - Improve the merge success rate of call logs on the agent side, significantly reducing the proportion of call logs with `response_status = Unknown`. In test environments, a 50% reduction in unknown proportion is observed. + - Limit the bandwidth consumption of data sent by the agent, with a default allowance of 100Mbps, [Documentation](../configuration/agent/#global.communication.max_throughput_to_ingester). + +# v7.0.5 [2025/04/02] + +## Stable Feature + +- AutoTracing + - Support collection and tracing of RocketMQ protocol, [Documentation](../features/l7-protocols/mq/#rocketmq). + - Support collection and tracing of Ping protocol, [Documentation](../features/l7-protocols/network/#ping). + - Support collection and tracing of Dubbo protocol when using Fastjson serialization, [Documentation](../features/l7-protocols/rpc/#dubbo). + +# v7.0.4 [2025/03/18] + +## Stable Feature + +- Agent + - Support collecting traffic from Pod internal NICs, applicable to scenarios where Pod NIC traffic cannot be directly collected under the Root network namespace (e.g., [Huawei Cloud CCE Turbo CNI](https://support.huaweicloud.com/usermanual-cce/cce_10_0284.html)), [Documentation](../configuration/agent/#inputs.cbpf.af_packet.inner_interface_capture_enabled). + +# v7.0.3 [2025/03/05] + +N/A + +# v7.0.2 [2025/02/11] + +## Stable Feature + +- AutoMetrics + - Add timeout ratio metric (`timeout_ratio`) to application performance metrics (`application`, `application_map`). +- Server + - Support terminating remote upgrades of collectors and optimize CPU resource usage of the Server during upgrades. +- Agent + - Support limiting the number of sockets used by deepflow-agent, [Documentation](../configuration/agent/#global.limits.max_sockets). + +# v7.0.1 [2025/01/16] + +## Stable Feature + +- AutoTracing + - For non-TCP traffic in network flow logs (`l4_flow_log`), change the end status (`close_type`) from timeout to normal end (1). + +# v7.0.0 [2025/01/02] + +## Stable Feature + +- AutoTracing + - Support collection and tracing of Tars protocol, [Documentation](../features/l7-protocols/rpc/#tars). \ No newline at end of file diff --git a/translate/translated/10-release-notes/06-ee-7.0-release.md b/translate/translated/10-release-notes/06-ee-7.0-release.md new file mode 100644 index 00000000..9a75e743 --- /dev/null +++ b/translate/translated/10-release-notes/06-ee-7.0-release.md @@ -0,0 +1,118 @@ +--- +title: v7.0 EE Release Notes +permalink: /release-notes/release-7.0-ee +--- + +> This document was translated by ChatGPT + +# Business and Applications + +## Business Observability + +- (New UI) Added automatic correlation display of system metrics, events, and application logs in the right-slide page. +- Usability improvements + - Added `Metrics Analysis` capability to the right-slide page for quick classification and comparison of application and network performance metrics. + - (New UI) Optimized the usability of business definition operations, supporting drag-and-drop to define services in the topology. + +## Application Observability + +- AutoTracing + - ⭐ Support for RocketMQ protocol collection and tracing, [Documentation](../features/l7-protocols/mq/#rocketmq). + - ⭐ Support for Tars protocol collection and tracing, [Documentation](../features/l7-protocols/rpc/#tars). + - Support for Ping protocol collection and tracing, [Documentation](../features/l7-protocols/network/#ping). + - Support for Dubbo protocol collection and tracing when using Fastjson serialization, [Documentation](../features/l7-protocols/rpc/#dubbo). + - Support for parsing compressed MySQL calls, [Documentation](../configuration/agent/#processors.request_log.application_protocol_inference.protocol_special_config.mysql.decompress_payload). + - Support for parsing MySQL Login Response statements and truncated MySQL protocol content. + - Optimized parsing of unary-type gRPC calls, [Documentation](../configuration/agent/#processors.request_log.application_protocol_inference.protocol_special_config.grpc.streaming_data_enabled). + - Support for parsing multiple DNS requests in TCP Payload, and parsing SRV-type DNS call logs, [Documentation](https://en.wikipedia.org/wiki/SRV_record). + - Support for collecting Unix Socket call logs, and automatic tracing between TCP/UDP Socket call logs and Unix Socket call logs. + - Enriched eBPF hook points for file read/write event collection to improve adaptability. + - Support for parsing TraceID and SpanID from Borui and Yunzhihui APM. + - Support for cross-thread analysis of the parent Span of the current Span (system Span at the client process location). +- AutoMetrics + - Added timeout ratio metric (`timeout_ratio`) to application performance metrics (`application`, `application_map`). +- AutoTagging + - ⭐ Optimized the meaning of the `process_kname` field in call logs and file read/write event data, changing from `kernel thread name` to `system process` name for better readability. + - ⭐ Optimized the meaning of the `response_status` field in call logs and improved page prompt information. + - **Normal**: Response code is normal. + - **Client Error**: Response code indicates a client-side error, e.g., HTTP 4XX. + - **Server Error**: Response code indicates a server-side error, e.g., HTTP 5XX. + - **Timeout**: If no response is collected within a certain time, the request is marked as timed out. + - Collector `Application Session Merge Timeout Setting`: DNS and TLS default 15s, other protocols default 120s, [Documentation](../configuration/agent/#processors.request_log.timeouts.session_aggregate). + - **Unknown**: When concurrent requests exceed the collector's cache capacity, the oldest requests are marked as unknown. + - Collector `Session Aggregate Max Entries` setting: Default cache of 64K requests, [Documentation](../configuration/agent/#processors.request_log.tunning.session_aggregate_max_entries). + - **Parse Failed**: Response was collected but the response code could not be parsed due to truncation or compression. + - Collector `Payload Truncation` setting: Default parses the first 1024 bytes of Payload, [Documentation](../configuration/agent/#processors.request_log.tunning.payload_truncation). + +## Code Observability + +- Usability improvements + - Collect `Java/Python` OnCPU profiling data by default. + - Collect `deepflow-*` OnCPU profiling data by default. + +# Infrastructure + +## Asset Observability + +- ⭐ Added asset observability feature, supporting viewing observability data from the perspective of cloud hosts and container resources. + +## Network Observability + +- Changed the end status (`close_type`) of non-TCP traffic in network flow logs from timeout to normal end (1). +- Changed the default unit for all traffic rates on the page from bytes per second (`Bps`) to bits per second (`bps`). + +## Traffic Distribution + +- Distribution strategy supports specifying collector groups. + +# Customization + +## Dashboards + +- When using PromQL queries, support setting metric aliases, units, and thresholds. + +# Others + +## Resource List + +- AutoTagging + - Process resources + - ⭐ Automatically record gprocess name as jar/py file name to avoid all showing as java/python. + - Aggregate processes with the same `cmdline` within the same cloud host or the same K8s workload into a unique gprocess to reduce redundant process information. + - Optimized default values for process matcher, [Documentation](../configuration/agent/#inputs.proc.process_matcher). + - By default, ignore collection of `sleep/sh/bash/pause/runc` process information. + - By default, collect process information for `Java/Python`. + - By default, collect process information for `deepflow-*`. + - By default, collect process information in containers. + - Support for collecting and associating change events of K8s resource definitions and ConfigMaps. +- Usability improvements + - ⭐ Performance: Added KV search capability to list pages to improve search experience in large-scale resource scenarios with millions of entries. + - Added ID column to VPC resource list to align with cloud platforms. + - When entering a peering connection, VPC can be left empty to establish peering with all VPCs under the specified cloud platform. + +## System Management + +- Server + - ⭐ Support for MCP Server. + - ⭐ Support for defining indexes for fields such as attribute.X, metrics.X to speed up retrieval of commonly used fields. + - Support for terminating remote upgrades of collectors and optimizing CPU resource usage of Server during upgrades. + - Support for setting maximum query duration to avoid excessive resource consumption for large time-scale queries. +- Agent + - ⭐ OneAgent: Support for using deepflow-agent to collect application logs, host system metrics, and K8s container system metrics. + - ⭐ OneAgent: Support for using deepflow-agent for continuous probing. + - ⭐ Security: Support for limiting the number of Sockets used by deepflow-agent, [Documentation](../configuration/agent/#global.limits.max_sockets). + - ⭐ Adaptability: Support for collecting traffic from Pod internal NICs, suitable for scenarios where Pod NIC traffic cannot be directly collected under the Root network namespace (e.g., [Huawei Cloud CCE Turbo CNI](https://support.huaweicloud.com/usermanual-cce/cce_10_0284.html)), [Documentation](../configuration/agent/#inputs.cbpf.af_packet.inner_interface_capture_enabled). + - ⭐ Performance: Support for compressed transmission of PCAP data, with compression ratio up to 5:1 ~ 10:1, [Documentation](../configuration/agent/#outputs.compression.pcap). + - ⭐ Performance: Support for compressed sending of call logs and flow logs, with call log compression ratio up to 8:1 in test environments, [Documentation](../configuration/agent/#outputs.compression.l7_flow_log). + - ⭐ Performance: Optimized memory usage of Cache for application performance metrics in Agent by timely cleaning up expired LRU entries, reducing overall memory consumption by **43%** in test environments. + - ⭐ Performance: Aggregate and store flow logs generated by LB health checks, reducing flow log storage overhead by nearly **50%** in a production environment, [Documentation](../configuration/agent/#outputs.flow_log.aggregators.aggregate_health_check_l4_flow_log). + - ⭐ Performance: Improved call log merge success rate on the agent side, significantly reducing the proportion of `response_status = Unknown` call logs, with a 50% reduction in unknown rate observed in test environments. + - ⭐ Support for collecting virtual and physical NIC traffic on non-Open vSwitch DPDK KVM hosts, [Documentation](../configuration/agent/#inputs.ebpf.socket.uprobe.dpdk.command). + - Adapted to K8s CNI with identical MAC addresses for virtual NICs on the same host. + - Optimized resource overhead protection mechanism when application protocol recognition fails to avoid mistakenly disabling application protocol parsing, [Documentation](../configuration/agent/#processors.request_log.application_protocol_inference.inference_max_retries). + - Collector list supports displaying associated VPC information. + - Limit agent data sending bandwidth consumption, default allowing 100Mbps, [Documentation](../configuration/agent/#global.communication.max_throughput_to_ingester). + - When agent traffic reaches the rate limit, support choosing between `drop` or `wait` strategies; default is drop, can be configured to wait to improve data sending success rate, [Documentation](../configuration/agent/#global.communication.ingester_traffic_overflow_action). + - Added circuit breaker mechanism for free disk space in Agent runtime environment, [Documentation](../configuration/agent/#global.circuit_breakers.free_disk). + - Support for disabling Agent use of Swap memory, [Documentation](../configuration/agent/#global.tunning.swap_disabled). + - Optimization: Reduced work performed by Agent when in disabled state. \ No newline at end of file diff --git a/translate/translated/10-release-notes/07-ce-6.6-release.md b/translate/translated/10-release-notes/07-ce-6.6-release.md new file mode 100644 index 00000000..3fba6df6 --- /dev/null +++ b/translate/translated/10-release-notes/07-ce-6.6-release.md @@ -0,0 +1,254 @@ +--- +title: v6.6 CE Release Notes +permalink: /release-notes/release-6.6-ce +--- + +> This document was translated by ChatGPT + +# Backport From 7.0 + +- AutoTracing + - [2025/01/02] Support collection and tracing of the Tars protocol, [documentation](../features/l7-protocols/rpc/#tars). + - [2025/01/16] For non-TCP traffic in network flow logs (`l4_flow_log`), change the end status (`close_type`) from timeout to normal end (1). + - [2025/04/02] Support collection and tracing of the Ping protocol, [documentation](../features/l7-protocols/network/#ping). + - [2025/04/02] Support collection and tracing of the Dubbo protocol when using Fastjson serialization, [documentation](../features/l7-protocols/rpc/#dubbo). + - [2025/04/15] Support parsing MySQL Login Response statements. + - [2025/04/15] Support parsing multiple DNS requests in a TCP Payload. + - [2025/04/28] Enrich eBPF hook points for collecting file read/write events (`io_event`) to improve adaptability. + - [2025/05/29] Support collecting Unix Socket call logs (`l7_flow_log`) and automatic tracing between TCP/UDP Socket call logs and Unix Socket call logs. + - [2025/05/29] Support parsing SRV type DNS call logs, [documentation](https://en.wikipedia.org/wiki/SRV_record). + - [2025/05/29] Support parsing truncated MySQL protocol content. +- AutoTagging + - [2025/04/28] Optimize the meaning of the `process_kname` field in call logs and file read/write event data, changing from `kernel thread name` to `system process` name for better readability. + - [2025/04/28] Aggregate processes with the same `cmdline` within the same cloud host or the same K8s workload into a unique gprocess to reduce redundant process information. + - [2025/04/28] Optimize default values for the process matcher, [documentation](../configuration/agent/#inputs.proc.process_matcher). + - By default, ignore collection of process information for `sleep/sh/bash/pause/runc`. + - By default, collect process information and OnCPU profiling data for `Java/Python`, and automatically record the gprocess name as the jar/py file name to avoid all being displayed as java/python. + - By default, collect process information and OnCPU profiling data for `deepflow-*`. + - By default, collect process information in containers. + - [2025/04/28] Optimize the meaning of the `response_status` field in call logs (`l7_flow_log`). + - **Normal**: Response code is normal. + - **Client Error**: Response code indicates a client-side error, e.g., HTTP 4XX. + - **Server Error**: Response code indicates a server-side error, e.g., HTTP 5XX. + - **Timeout**: If no response is collected within a certain time, the request is marked as timed out. + - Agent `Application session merge timeout` configuration: DNS and TLS default 15s, other protocols default 120s, [documentation](../configuration/agent/#processors.request_log.timeouts.session_aggregate). + - **Unknown**: When concurrent requests exceed the collector's cache capacity, the oldest requests are marked as unknown. + - Agent `Maximum session aggregation entries` configuration: Default cache of 64K requests, [documentation](../configuration/agent/#processors.request_log.tunning.session_aggregate_max_entries). + - **Parse Failed**: Response was collected but the response code could not be parsed due to truncation or compression. + - Agent `Payload truncation` configuration: Default parses the first 1024 bytes of the Payload, [documentation](../configuration/agent/#processors.request_log.tunning.payload_truncation). + - [2025/06/11] Optimize parsing of unary type gRPC calls, [documentation](../configuration/agent/#processors.request_log.application_protocol_inference.protocol_special_config.grpc.streaming_data_enabled). + - [2025/08/21] Support collecting multiple HTTP2/gRPC requests and responses in a single packet. + - [2025/08/21] Support obtaining the full file path for file read/write events: + - Fully obtain NAS file paths, supporting NFS, SMB, CIFS, and other protocols. + - Fully obtain the absolute path for file read/write inside container Pods. +- AutoMetrics + - [2025/08/21] Support aggregation to generate eBPF profiling metric data with 1s granularity to speed up profiling metric queries. +- AutoTagging + - [2025/08/21] Simplify process sync blacklist configuration, [documentation](../configuration/agent/#inputs.proc.process_blacklist). + - [2025/08/21] Adapt to K8s v1.32+ API. +- Server + - [2025/02/11] Support terminating remote upgrades of collectors and optimize CPU resource usage of the Server during upgrades. +- Agent + - [2025/02/11] Support limiting the number of sockets used by deepflow-agent, [documentation](../configuration/agent/#global.limits.max_sockets). + - [2025/03/18] Support collecting traffic from Pod internal NICs, applicable to scenarios where Pod NIC traffic cannot be directly collected under the Root network namespace (e.g., [Huawei Cloud CCE Turbo CNI](https://support.huaweicloud.com/usermanual-cce/cce_10_0284.html)), [documentation](../configuration/agent/#inputs.cbpf.af_packet.inner_interface_capture_enabled). + - [2025/04/15] Limit the bandwidth consumption of data sent by the agent, default allowing 100Mbps, [documentation](../configuration/agent/#global.communication.max_throughput_to_ingester). + - [2025/04/28] Optimize memory usage of the cache for application performance metrics in the Agent by timely cleaning up expired LRU entries, reducing overall memory consumption by **43%** in test environments. + - [2025/04/28] Aggregate and store flow logs (`l4_flow_log`) generated by LB health checks, reducing flow log storage overhead by nearly **50%** in some production environments, [documentation](../configuration/agent/#outputs.flow_log.aggregators.aggregate_health_check_l4_flow_log). + - [2025/04/28] Optimize resource overhead protection mechanism when application protocol recognition fails to avoid mistakenly disabling application protocol parsing, [documentation](../configuration/agent/#processors.request_log.application_protocol_inference.inference_max_retries). + - [2025/05/16] Support compressed transmission of call logs and flow logs, with a compression ratio of up to 8:1 in test environments, [documentation](../configuration/agent/#outputs.compression.l7_flow_log). + - [2025/05/29] When Agent traffic reaches the rate limit, support choosing between `drop` or `wait` strategies; default is drop, can be configured to wait to improve data transmission success rate, [documentation](../configuration/agent/#global.communication.ingester_traffic_overflow_action). + - [2025/06/11] Add a circuit breaker mechanism for free disk space in the Agent runtime environment, [documentation](../configuration/agent/#global.circuit_breakers.free_disk). + - [2025/06/11] Support disabling Agent's use of swap memory, [documentation](../configuration/agent/#global.tunning.swap_disabled). + - [2025/06/11] Adapt to K8s CNI with identical MAC addresses for virtual NICs on the same host. + - [2025/06/11] Optimization: Reduce work done by the Agent when disabled. + - [2025/08/21] Support Watchdog mechanism to ensure circuit breakers execute properly in extreme cases. + - [2025/08/21] Support compressed transmission of application logs, with compression ratios between 5:1 and 20:1, [documentation](../configuration/agent/#outputs.compression.application_log). + +# v6.6.9 [2024/12/12] + +## Stable Feature + +- AutoTracing + - Support collection and tracing of the Memcached protocol, [documentation](../features/l7-protocols/nosql/#memcached). + - cBPF data supports Tars protocol parsing, [documentation](../features/l7-protocols/rpc/#tars). + - File read/write events support collecting the full path of file names and the offset of read/write files. +- AutoProfiling + - Support CPU performance profiling for Python and CUDA. + - Optimize Java process symbol table synchronization mechanism, reducing transient CPU consumption introduced to business processes by about 50%. + - Improve function stack merging efficiency, reducing resource overhead for function stack reporting, with significant performance improvement in scenarios with many threads of the same name. +- AutoTagging + - When TraceID exists in the protocol header, support disabling eBPF syscall_trace_id calculation (via `syscall_trace_id_disabled`) to reduce impact on business performance. + - Support completely disabling cBPF data collection (by setting `tap_interface_regex` to an empty string) to reduce memory overhead. + - Enhance process synchronization capability, [documentation](../configuration/agent/#inputs.proc.process_matcher). + - Support synchronizing only processes inside containers. + - Support not synchronizing Socket information (only process information). + - When a region whitelist is configured for the cloud platform (Domain), calling the Region API is no longer required. + - Failure to obtain NAT gateway, routing table, or load balancer information from Alibaba Cloud or Tencent Cloud will not affect synchronization of other resource information. +- Server + - Optimize storage performance of `genesis*` related MySQL tables. + - Support using ByConity instead of ClickHouse, [documentation](../best-practice/storage-engine-use-byconity/). + - Support using ClickHouse Enterprise Edition (currently only supported on Alibaba Cloud), [documentation](https://www.aliyun.com/product/apsaradb/clickhouse). +- Agent + - Support compressed transmission of profiling data, reducing bandwidth consumption by 30%. + - Application log data supports compressed transmission, reducing bandwidth consumption by 95% (CPU consumption increases by 3%). + - Support deepflow-agent using a single socket to transmit all observability data, and allow disabling this feature via `multiple_sockets_to_ingester` to use multiple sockets for improved transmission performance. + - When BTF (BPF Type Format) is enabled on Linux, and the kernel is >= [5.5](https://github.com/torvalds/linux/commit/f1b9509c2fb0ef4db8d22dac9aef8e856a5d81f6) on X86 architecture or >= [6.0](https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git/commit/?h=linux-6.0.y&id=efc9909fdce00a827a37609628223cd45bf95d0b) on ARM architecture, the agent will automatically use fentry/fexit instead of kprobe/kretprobe, resulting in about 15% performance improvement. + - The original environment variable `ONLY_WATCH_K8S_RESOURCE` has been replaced with `K8S_WATCH_POLICY`, [documentation](../ce-install/serverless-pod/#部署-deepflow-agent). + +# v6.6.8 [2024/11/14] + +## Stable Feature + +- Server + - By default, aggregate and generate network performance metrics and application performance metrics with granularity of 1h and 1d. +- Agent + - Configuration refactoring, [documentation](../configuration/agent/). + +# v6.6.7 [2024/10/31] + +## Beta Feature + +- AutoTagging + - Enhance process synchronization capability, [documentation](../configuration/agent/#inputs.proc.process_matcher). + - Support synchronizing only processes inside containers. + - Support not synchronizing Socket information (only process information). + +# v6.6.6 [2024/10/11] + +## Backward Incompatible Change + +- AutoTracing + - To reduce resource overhead and avoid misidentification, the agent will by default only parse the following application protocols (to enable parsing of other protocols, configure `l7-protocol-enabled`): + - HTTP, HTTP2/gRPC, MySQL, Redis, Kafka, DNS, TLS. + - Reminder: When using Wasm to parse private protocols, please add Custom to `l7-protocol-enabled`. + +## Stable Feature + +- Agent + - Support specifying and disabling K8s List & Watch via environment variables (thanks to `Hyzhou`: [FR](https://github.com/deepflowio/deepflow/issues/5404), [FR](https://github.com/deepflowio/deepflow/issues/7965)). + - Reduce eBPF memory overhead of the Agent (thanks to `qyzhaoxun`: [FR](https://github.com/deepflowio/deepflow/issues/8028)). + +# v6.6.5 [2024/09/24] + +## Beta Feature + +- AutoProfiling + - Optimize Java process symbol table synchronization mechanism, reducing transient CPU consumption introduced to business processes by about 50%. + - Improve function stack merging efficiency, reducing resource overhead for function stack reporting, with significant performance improvement in scenarios with many threads of the same name. +- Server + - Optimize storage performance of `genesis*` related MySQL tables. + - AutoTagging: When a region whitelist is configured for the cloud platform (Domain), calling the Region API is no longer required. + - AutoTagging: Failure to obtain NAT gateway, routing table, or load balancer information from Alibaba Cloud or Tencent Cloud will not affect synchronization of other resource information. +- Agent + - When BTF (BPF Type Format) is enabled on Linux, and the kernel is >= [5.5](https://github.com/torvalds/linux/commit/f1b9509c2fb0ef4db8d22dac9aef8e856a5d81f6) on X86 architecture or >= [6.0](https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git/commit/?h=linux-6.0.y&id=efc9909fdce00a827a37609628223cd45bf95d0b) on ARM architecture, the agent will automatically use fentry/fexit instead of kprobe/kretprobe, resulting in about 15% performance improvement. + - Support compressed transmission of profiling data, reducing bandwidth consumption by 30%. + - The original environment variable `ONLY_WATCH_K8S_RESOURCE` has been replaced with `K8S_WATCH_POLICY`, [documentation](../ce-install/serverless-pod/#部署-deepflow-agent). + +## Stable Feature + +- AutoTracing + - Support enhancing HTTP2/gRPC call logs using Wasm Plugin (currently not supporting enhancement of eBPF uprobe data), [documentation](../integration/process/wasm-plugin/). +- AutoProfiling + - Support stack unwinding using DWARF when Frame Pointer is missing. +- AutoTagging + - Support Alibaba Cloud resource synchronization using a regular account's AK/SK with ResourceGroupId. + +# v6.6.4 [2024/08/29] + +## Beta Feature + +- AutoTracing + - cBPF data supports Tars protocol parsing, [documentation](../features/l7-protocols/rpc/#tars). +- AutoProfiling + - Support stack unwinding using DWARF when Frame Pointer is missing. +- AutoTagging + - Support Alibaba Cloud resource synchronization using a regular account's AK/SK with ResourceGroupId. +- Server + - Support using ByConity instead of ClickHouse, [documentation](../best-practice/storage-engine-use-byconity/). + +## Stable Feature + +- AutoTracing + - Automatically correct minor clock drift between different machines in distributed tracing flame graphs. +- AutoTagging + - Support customizing K8s workload abstraction rules using Lua Plugin, [documentation](../features/auto-tagging/k8s-crd/). + - Support synchronizing LoadBalancer type container services. +- Server + - Support using OceanBase instead of MySQL. + +# v6.6.3 [2024/08/15] + +## Beta Feature + +- AutoTracing + - When TraceID exists in the protocol header, support disabling eBPF syscall_trace_id calculation (via `syscall_trace_id_disabled`) to reduce impact on business performance. + - Automatically correct minor clock drift between different machines in distributed tracing flame graphs. +- AutoTagging + - Support customizing K8s workload abstraction rules using Lua Plugin, [documentation](../features/auto-tagging/k8s-crd/). +- Agent + - Support completely disabling cBPF data collection (by setting `tap_interface_regex` to an empty string) to reduce memory overhead. + - Support deepflow-agent using a single socket to transmit all observability data, and allow disabling this feature via `multiple_sockets_to_ingester` to use multiple sockets for improved transmission performance. + +## Stable Feature + +- AutoProfiling + - Support viewing DeepFlow eBPF On-CPU Profiling data in Grafana Panel, [Demo](https://ce-demo.deepflow.yunshan.net/d/Continuous_Profiling/continuous-profiling?var-app_service=deepflow-server). +- AutoMetrics + - Support aligning timestamps of request and response metrics within the same session to help AIOps systems better perform root cause analysis (thanks to `pegasusljn`: [FR](https://github.com/deepflowio/deepflow/issues/7069)). +- AutoTagging + - Correctly tag Universal Tag for loopback NIC traffic on K8s Nodes. +- Agent + - Reduce the number of sockets used by deepflow-agent when sending data. + - Merge sockets used for transmitting open_telemetry and open_telemetry_compressed data when integrating with OpenTelemetry. + - Merge sockets used for agent self-monitoring, transmitting deepflow_stats and agent_log data. + - Merge sockets used for transmitting prometheus and telegraf metrics when integrating with Prometheus and Telegraf. + +# v6.6.2 [2024/08/01] + +## Beta Feature + +- AutoMetrics + - Support aligning timestamps of request and response metrics within the same session to help AIOps systems better perform root cause analysis (thanks to `pegasusljn`: [FR](https://github.com/deepflowio/deepflow/issues/7069)). + +## Stable Feature + +- AutoTracing + - Optimize default values for NTP clock offset (`host_clock_offset_us`) and network delay (`network_delay_us`) configuration parameters used in network span tracing to reduce mismatch probability. + +# v6.6.1 [2024/07/18] + +## Beta Feature + +- AutoTagging + - Correctly tag Universal Tag for loopback NIC traffic on K8s Nodes. + +## Stable Feature + +- AutoTracing + - Add URL masking capability for HTTP protocol, enable Redis protocol masking by default. +- AutoTagging + - Support synchronizing Volcano Engine resource tags, [documentation](../features/auto-tagging/meta-tags/). + - Stop synchronizing Pods in K8s Evicted state to reduce resource overhead. +- Integration + - Optimize mapping of schema/target and other fields in OTel Span to `l7_flow_log`, [documentation](../features/l7-protocols/otel/). +- Agent + - Support aggregated collection of traffic from multiple member physical NICs of an Open vSwitch bond interface. + +# v6.6.0 [2024/07/04] + +## Backward Incompatible Change + +- AutoProfiling + - Use Dataframe return format to compress response size and improve API performance, [PR](https://github.com/deepflowio/deepflow/pull/7011), [documentation](../features/continuous-profiling/data/). + +| | #Functions | Response Size (Byte) | Download Time | +| ------ | ---------- | -------------------- | ------------- | +| Before | 450,000 | 21.9M | 6.16s | +| After | 450,000 | 3.07M | 0.78s | + +## Beta Feature + +- AutoTagging + - Support synchronizing Volcano Engine resource tags, [documentation](../features/auto-tagging/meta-tags/). +- Agent + - Support aggregated collection of traffic from multiple member physical NICs of an Open vSwitch bond interface. \ No newline at end of file diff --git a/translate/translated/10-release-notes/08-ee-6.6-release.md b/translate/translated/10-release-notes/08-ee-6.6-release.md new file mode 100644 index 00000000..a8885f7b --- /dev/null +++ b/translate/translated/10-release-notes/08-ee-6.6-release.md @@ -0,0 +1,325 @@ +--- +title: v6.6 EE Release Notes +permalink: /release-notes/release-6.6-ee +--- + +> This document was translated by ChatGPT + +# Business and Applications + +## Business Observability + +- Data Association + - ⭐ The universal map now supports displaying alert events. +- Usability Enhancements + - Optimized topology style. + +## Application Observability + +- AutoTracing + - ⭐ Introduced TraceMap capability, which calculates the aggregate topology of all Traces matching the search criteria in real-time, helping users quickly organize software architecture and globally locate performance bottlenecks. + - ⭐ Supports using Wasm Plugin to enhance HTTP2/gRPC call logs (currently does not support enhancing eBPF uprobe data), [documentation](../integration/process/wasm-plugin/). + - ⭐ File read/write events now support collecting the full path of the file name and the offset of the read/write file. + - Supports collection and tracing of Memcached protocol, [documentation](../features/l7-protocols/nosql/#memcached). + - Supports collection and tracing of Tars protocol, [documentation](../features/l7-protocols/rpc/#tars). + - Supports collection and tracing of Ping protocol, [documentation](../features/l7-protocols/network/#ping). + - Supports collection and tracing of Dubbo protocol when using Fastjson serialization, [documentation](../features/l7-protocols/rpc/#dubbo). + - Supports parsing MySQL Login Response statements and parsing truncated MySQL protocol content. + - Optimized parsing of unary type gRPC calls, [documentation](../configuration/agent/#processors.request_log.application_protocol_inference.protocol_special_config.grpc.streaming_data_enabled). + - Supports parsing multiple DNS requests in TCP Payload and parsing SRV type DNS call logs, [documentation](https://en.wikipedia.org/wiki/SRV_record). + - Supports collection and tracing of Some/IP protocol. + - Introduced URL obfuscation capability for HTTP protocol, with Redis protocol obfuscation enabled by default, [documentation](../configuration/agent/#processors.request_log.tag_extraction.obfuscate_protocols). + - Optimized default values for NTP clock offset (`host_clock_offset_us`) and network transmission delay (`network_delay_us`) configuration parameters used in network Span tracing to reduce the probability of mismatches. + - Optimized mapping of schema/target fields in OTel Span to `l7_flow_log`, [documentation](../features/l7-protocols/otel/). + - Supports collecting Unix Socket call logs and automatic tracing between TCP/UDP Socket call logs and Unix Socket call logs. + - Enriched eBPF Hook points for file read/write event collection to enhance adaptability. + - Supports parsing TraceID and SpanID from APMs like BoRui, Tingyun, and Cloudwise. + - Supports cross-thread analysis of the parent Span of the current Span (system Span at the client process location). + - Supports collecting multiple HTTP2/gRPC requests and responses in a single Packet. + - Supports obtaining the full file path for file read/write events: + - Can fully obtain NAS file paths, supporting NFS, SMB, CIFS, and other protocols. + - Can fully obtain the complete absolute path of file reads/writes inside a container Pod. +- AutoMetrics + - Supports aligning timestamps of request and response metrics within the same session to help AIOps systems better achieve root cause localization. +- AutoTagging + - Supports aggregating traffic of multiple member physical network cards of Open vSwitch bond interface, [documentation](../configuration/agent/#inputs.cbpf.af_packet.bond_interfaces). + - Correctly marks Universal Tag for loopback network card traffic on K8s Node. + - ⭐ Optimized the meaning of the `process_kname` field in call logs and file read/write event data from `kernel thread name` to `system process` name for better readability. + - ⭐ Optimized the meaning of the response status (`response_status`) field in call logs and improved page prompt information. + - **Normal**: Response code is normal. + - **Client Error**: Response code indicates a client-side error, such as HTTP 4XX. + - **Server Error**: Response code indicates a server-side error, such as HTTP 5XX. + - **Timeout**: If no response is collected within a certain time, the request is marked as timed out. + - Collector `application session merge timeout setting` configuration: DNS and TLS default to 15s, other protocols default to 120s, [documentation](../configuration/agent/#processors.request_log.timeouts.session_aggregate). + - **Unknown**: When concurrent request volume exceeds the collector's cache capacity, the oldest request is marked as unknown. + - Collector `session aggregation maximum entries` configuration: Default cache of 64K requests, [documentation](../configuration/agent/#processors.request_log.tunning.session_aggregate_max_entries). + - **Parse Failure**: A response was collected, but due to truncation or compression, the response code could not be parsed. + - Collector `Payload truncation` configuration: Default parses the first 1024 bytes of Payload, [documentation](../configuration/agent/#processors.request_log.tunning.payload_truncation). +- Search Capability + - Introduced `application search` mode, allowing quick selection of services (`app_service`) and instances (`app_instance`), and quick input of endpoints (`endpoint`) and TraceID. + - Optimized the layout of the search box and the display style of the quick selection box. +- Performance Enhancement + - Optimized the performance of distributed tracing API. +- Usability Enhancements + - ⭐ Added `backend analysis` capability to distributed tracing, intelligently guiding users to quickly trace requests to the backend when eBPF AutoTracing is disconnected. + - ⭐ Supports quickly viewing continuous profiling and real-time profiling data corresponding to system Span. + - ⭐ Distributed tracing supports `waterfall list` display mode. + - ⭐ deepflow-agent supports directly receiving tracing data from SkyWalking and Datadog without forwarding through otel-collector. + - ⭐ Distributed tracing flame graph automatically corrects slight clock deviations between different machines. + - ⭐ Optimized the usability of the `network path` right slide page. + - Optimized the process configuration capability for enabling eBPF uprobe functionality, [documentation](../configuration/agent/#inputs.proc.process_matcher). + - Optimized the distributed tracing page: adjusted the order of the quick filter box on the left, added trend analysis line charts, and optimized the call log details table. + - Resource analysis, path analysis, and topology analysis pages support automatically selecting the most appropriate metric time granularity for queries to optimize query speed. + - Resource analysis, path analysis, and topology analysis pages support quickly filtering the value range of metrics. + - The call log in the right slide page only displays abnormal entries by default. + - Optimized the display position of Tips in the topology map. + - Optimized the display of tag classification in both Chinese and English. + - Enriched the quick filter box on the left side of the page. + - Optimized the display order of search box candidates. + +## Code Observability + +- AutoProfiling + - ⭐ Supports eBPF zero-intrusion collection of memory profiling data for Java and Rust processes, [documentation](../configuration/agent/#inputs.ebpf.profile.memory). + - ⭐ Supports using DWARF to achieve stack unwinding in the absence of Frame Pointer, [documentation](../configuration/agent/#inputs.ebpf.profile.unwinding). + - ⭐ Supports CPU performance profiling for Python and CUDA. +- AutoMetrics + - Supports aggregating and generating eBPF profiling metric data with 1s granularity to accelerate profiling metric queries. +- Performance Enhancement + - ⭐ Optimized Java process symbol table synchronization mechanism, reducing instantaneous CPU consumption introduced by business processes by about 50%. + - ⭐ Improved function stack merging efficiency, reducing resource overhead for function stack reporting, with significant performance improvement in scenarios with many threads of the same name. + - ⭐ Supports compressed transmission of Profiling data, reducing bandwidth consumption by 30%. + - ⭐ Supports compressed transmission of call logs and flow logs, with a compression ratio of up to 8:1 in test environments, [documentation](../configuration/agent/#outputs.compression.l7_flow_log). +- Real-time Profiling + - ⭐ Supports using JVM API to obtain function stack, GC statistics, and Heap statistics of Java processes. +- Grafana + - Supports viewing DeepFlow eBPF On-CPU Profiling data in Grafana Panel. +- Usability Enhancements + - Optimized the process configuration capability for enabling Profile functionality, [documentation](../configuration/agent/#inputs.proc.process_matcher). + - Optimized the display of function types in flame graphs. + - Optimized the text of memory profiling flame graphs. + - Enriched the quick filter box on the left side of the page. + - Optimized the display order of search box candidates. + - Default collection of OnCPU profiling data for `Java/Python`. + - Default collection of OnCPU profiling data for `deepflow-*`. + +# Infrastructure + +## Asset Observability + +- ⭐ Introduced asset observability feature, supporting viewing observability data from the perspective of cloud host and container resources. + +## Network Observability + +- AutoTagging + - Supports aggregating traffic of multiple member physical network cards of Open vSwitch bond interface, [documentation](../configuration/agent/#inputs.cbpf.af_packet.bond_interfaces). + - Correctly marks Universal Tag for loopback network card traffic on K8s Node. +- PCAP + - ⭐ Supports online analysis of PCAP packet data. +- Performance Enhancement + - Optimized NAT tracing API performance. +- Usability Enhancements + - ⭐ Optimized the usability of the `network path` right slide page. + - Changed the default unit for all traffic rates on the page from bytes per second (`Bps`) to bits per second (`bps`). + - For non-TCP traffic in network flow logs (`l4_flow_log`), changed the end status (`close_type`) from timeout to normal end (1). + - Resource analysis, path analysis, and topology analysis pages support automatically selecting the most appropriate metric time granularity for queries to optimize query speed. + - Resource analysis, path analysis, and topology analysis pages support quickly filtering the value range of metrics. + - Traffic of `collector=other network cards` is included in resource metrics. + - Supports aggregation of tunnel and non-tunnel traffic, solving the problem of asymmetric path traffic aggregation. + - When collecting TCP packet headers or PCAP data, network flow log collection is automatically enabled. + - Optimized the display of information bar on the NAT tracing page. + - Optimized the display position of Tips in the topology map. + - The flow log in the right slide page only displays abnormal entries by default. + - Optimized the display of tag classification in both Chinese and English. + - Provides graphical interpretation of flow log end types (`close_type`). + - Enriched the quick filter box on the left side of the page. + - Optimized the display order of search box candidates. + +## Traffic Distribution + +- ⭐ Traffic distribution supports ZMQ protocol. +- Distribution strategy supports specifying collector groups. + +# Integration + +## Probing Center + +- Real-time Probing + - ⭐ deepflow-agent has built-in probing capability, eliminating the need to install binary files for probing commands. + - Supported commands include: ping, tcpping, curl, dig, traceroute. + - Supports executing probing commands within business Pods. + +# Customization + +## Dashboard + +- Panel Enhancements + - ⭐ Aggregated metrics and overview charts support PromQL queries to enhance the display capability of Prometheus metrics in the dashboard. + - Optimized the display of Tips in Panels, showing query names when there are multiple query conditions, and compactly (ignoring) displaying metric names when there is only a single metric. + - Topology maps support setting thresholds for the difference in path metrics (between adjacent hops). + - Tables support more extensive color settings. + - Panels support setting the color of legends. + - Optimized the customization capability of pie charts. + - Added the ability to zoom in for a closer look. + - Unified the method for setting metric units. +- Usability Enhancements + - Simplified the method for setting aliases, units, and thresholds for Panel metrics. + - Optimized the right slide detail page for resource change events and file read/write events. + - Optimized the display of Tab pages in the right slide page. + - Supports adding tags to dashboards. + - Supports copying entire dashboards. + - Optimized legend display. + +# Integration + +## Log Center + +- Performance Enhancement + - ⭐ Application log data supports compressed transmission, reducing bandwidth consumption by 95% (CPU consumption increased by 3%). +- Usability Enhancements + - Optimized the search box, fixing the selection of application service (`app_service`) filter condition. + +# Others + +## Usability + +- Supports copying and pasting the entire content of the search box. + +## Alert Management + +- Alert Strategy + - ⭐ Email push content supports Markdown format and supports Jinja2 syntax for referencing tags. + - The search module supports setting metric aliases and viewing units. + - Supports customizing the criteria for alert recovery: if there are no "fatal/error/warning/no data" monitoring events for N consecutive times, a recovery event is generated. +- Alert Events + - ⭐ Enriched alert event Tags to align with all observability data. + - Added `event analysis` page for statistical analysis of alert events. +- Push Endpoints + - When pushing to Kafka, supports `SCRAM-SHA-256` authentication method. +- Usability Enhancements + - Optimized system alert events to display detailed internal module names for DeepFlow process anomalies. + - Optimized the display of time range in the alert event right slide box. +- Performance Enhancement + - Optimized page loading time. + +## Report Management + +N/A + +# Management + +## Resource List + +- AutoTagging + - ⭐ Supports synchronizing resource tags from Volcengine, [documentation](../features/auto-tagging/meta-tags/). + - Supports synchronizing LoadBalancer type container services. + - Enhanced process synchronization capability, [documentation](../configuration/agent/#inputs.proc.process_matcher). + - ⭐ Supports synchronizing only processes inside containers. + - Supports not synchronizing Socket information (only synchronizing process information). + - Process Resources + - ⭐ Automatically records gprocess name as jar/py file name to avoid all displaying as java/python. + - Aggregates processes with the same `cmdline` on the same cloud host or the same K8s workload into a unique gprocess, reducing redundant process information. + - Optimized default values for process matchers, [documentation](../configuration/agent/#inputs.proc.process_matcher). + - By default, ignores the collection of `sleep/sh/bash/pause/runc` process information. + - By default, collects process information for `Java/Python`. + - By default, collects process information for `deepflow-*`. + - By default, collects process information inside containers. + - Supports entering custom tags for cloud.tag on the page. + - Simplified process synchronization blacklist configuration, [documentation](../configuration/agent/#inputs.proc.process_blacklist). + - Adapted to K8s v1.32+ API. +- Management Capability + - When entering a K8s cluster, supports specifying ClusterID to reuse the old ClusterID when re-entering the cluster. + - Supports using Lua Plugin to customize K8s workload abstraction rules, [documentation](../features/auto-tagging/k8s-crd/). + - Limits to only one `collector synchronization` type cloud platform per organization per region. + - Supports completing Alibaba Cloud resource synchronization using a regular account's AK/SK with ResourceGroupId. + - Predefined system alerts for resource synchronization lag and resource relationship anomalies. +- Performance Enhancement + - Cancels synchronization of K8s Pods in Evicted state to reduce resource overhead. + - Optimized storage performance of `genesis*` related MySQL tables. +- Adaptability Optimization + - When a cloud platform (Domain) is configured with a region whitelist, there is no need to call the Region API. + - Failure to obtain NAT gateways, route tables, and load balancers for Alibaba Cloud and Tencent Cloud does not affect the synchronization of other resource information. + +## System Management + +- Server + - ⭐ Supports using OceanBase to replace MySQL. + - ⭐ Supports using ByConity to replace ClickHouse, [documentation](../best-practice/storage-engine-use-byconity/). + - ⭐ Supports using ClickHouse Enterprise Edition (currently only supports Alibaba Cloud), [documentation](https://www.aliyun.com/product/apsaradb/clickhouse). + - Supports terminating remote upgrades of collectors, optimizing CPU resource overhead during upgrades. + - Default aggregation generates network performance metrics and application performance metrics with granularity of 1h and 1d. + - Filters for data export (Kafka/Prometheus/OTel) (`tag-filters-groups`) support filling in multiple groups to achieve logical OR semantics. + - Unified log format for each module in deepflow-server. + - Supports setting the maximum query duration to avoid excessive resource consumption for large time-scale queries. + - Performance Enhancement + - ClickHouse accesses MySQL through a proxy to obtain dictionary data, reducing MySQL connections and optimizing cross-region bandwidth consumption. +- Agent + + - ⭐ OneAgent: Supports using deepflow-agent to collect application logs, host system metrics, and K8s container system metrics. + - ⭐ OneAgent: Supports using deepflow-agent for continuous probing. + - ⭐ Security: Supports limiting the number of Sockets used by deepflow-agent, [documentation](../configuration/agent/#global.limits.max_sockets). + - ⭐ Configuration refactoring, significantly improving usability, [documentation](../configuration/agent/). + - ⭐ Supports collecting traffic of virtual and physical network cards on DPDK KVM hosts that are not Open vSwitch, [documentation](../configuration/agent/#inputs.ebpf.socket.uprobe.dpdk.command). + - Supports collecting traffic of internal network cards in Pods, suitable for scenarios where Pod network card traffic cannot be directly collected under the Root network namespace (e.g., [Huawei Cloud CCE Turbo CNI](https://support.huaweicloud.com/usermanual-cce/cce_10_0284.html)), [documentation](../configuration/agent/#inputs.cbpf.af_packet.inner_interface_capture_enabled). + - Adapted to K8s CNI with the same MAC address for virtual network cards on the same host. + - Supports specifying and disabling K8s List & Watch through environment variables, [documentation](../ce-install/serverless-pod). + - Supports decapsulation of VXLAN type remote mirror traffic, [documentation](../configuration/agent/#inputs.cbpf.preprocess.tunnel_trim_protocols). + - Dedicated collectors support setting to ignore PCP processing of mirror traffic, [documentation](../configuration/agent/#inputs.cbpf.af_packet.vlan_pcp_in_physical_mirror_traffic). + - Dedicated collectors support calculating network location (capture_network_type) based on QinQ inner VLAN, [documentation](../configuration/agent/#inputs.cbpf.af_packet.vlan_pcp_in_physical_mirror_traffic). + - Dedicated collectors by default do not limit the number of concurrent flows and the memory overhead of the policy module, [documentation](../configuration/agent/#processors.flow_log.tunning.concurrent_flow_limit). + - Idle memory circuit breaker mechanism supports using available memory metrics, [documentation](../configuration/agent/#global.circuit_breakers.sys_memory_percentage.metric). + - Limits the bandwidth consumption of data sent by the agent, allowing 100Mbps of data to be sent by default, [documentation](../configuration/agent/#global.communication.max_throughput_to_ingester). + - When the agent's traffic reaches the rate limit, it supports choosing between `discard` or `wait` strategies, with the default behavior being discard, configurable to wait to improve data transmission success rate, [documentation](../configuration/agent/#global.communication.ingester_traffic_overflow_action). + - Optimized resource overhead protection mechanism when application protocol recognition fails to avoid mistakenly prohibiting application protocol parsing, [documentation](../configuration/agent/#processors.request_log.application_protocol_inference.inference_max_retries). + - Introduced a circuit breaker mechanism for the disk free space of the Agent runtime environment, [documentation](../configuration/agent/#global.circuit_breakers.free_disk). + - Supports prohibiting the Agent from using Swap memory, [documentation](../configuration/agent/#global.tunning.swap_disabled). + - Optimization: Reduced the work done by the Agent in a disabled state. + - Reduced the number of Sockets used by deepflow-agent when sending data: + - Merged Sockets used for transmitting open_telemetry and open_telemetry_compressed data when integrating OpenTelemetry. + - Merged Sockets used for agent self-monitoring, transmitting deepflow_stats and agent_log data. + - Merged Sockets used for transmitting prometheus and telegraf metrics when integrating Prometheus and Telegraf. + - Performance Enhancement + + - ⭐ Reduced eBPF kernel memory overhead of the Agent, reducing memory consumption by 60% under default configuration. + - ⭐ Supports using BPF FANOUT mechanism to improve collection performance, [documentation](../configuration/agent/#inputs.cbpf.af_packet.tunning.packet_fanout_count). + - ⭐ Performance: Optimized memory usage of Cache used for application performance metrics in the Agent by timely cleaning up expired LRU entries, reducing overall memory consumption by **43%** in test environments. + - ⭐ Performance: Aggregated storage of flow logs generated by LB health checks, reducing flow log storage overhead by nearly **50%** in a certain production environment, [documentation](../configuration/agent/#outputs.flow_log.aggregators.aggregate_health_check_l4_flow_log). + - ⭐ Performance: Improved the merge success rate of call logs on the agent side, significantly reducing the proportion of call logs with `response_status = Unknown`, with a 50% reduction in unknown proportion observed in test environments. + + - When TraceID is present in the protocol header, supports disabling eBPF syscall_trace_id calculation to reduce impact on business performance, [documentation](../configuration/agent/#inputs.ebpf.socket.tunning.syscall_trace_id_disabled). + - Supports completely disabling cBPF data collection (by configuring `inputs.cbpf.af_packet.interface_regex` to an empty string) to reduce memory overhead, [documentation](../configuration/agent/#inputs.cbpf.af_packet.interface_regex). + - Supports deepflow-agent using a single Socket to transmit all observability data, [documentation](../configuration/agent/#outputs.socket.multiple_sockets_to_ingester). + - When Linux has BTF (BPF Type Format) enabled, and the kernel is greater than or equal to [5.5](https://github.com/torvalds/linux/commit/f1b9509c2fb0ef4db8d22dac9aef8e856a5d81f6) on X86 architecture or greater than or equal to [6.0](https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git/commit/?h=linux-6.0.y&id=efc9909fdce00a827a37609628223cd45bf95d0b) on ARM architecture, the agent will automatically use fentry/fexit instead of kprobe/kretprobe, resulting in approximately 15% performance improvement. + + - Supports Watchdog mechanism to ensure circuit breakers can execute normally in extreme cases. + - Supports compressed transmission of application logs, with a compression ratio between 5:1 and 20:1, [documentation](../configuration/agent/#outputs.compression.application_log). + +- Usability Improvements + - Significantly reduced the URL length of web pages, remembering the activation status of the right slide box in the URL. + - Displays the container cluster to which the collector belongs in the collector list. + - Optimized the display of the navigation bar in both Chinese and English. + +## Account + +- Multi-Tenant Support + - ⭐ Supports setting visible pages, resources, databases/tables/fields/field enumeration values for tenants. + - ⭐ Supports specifying the team to which a specific affiliated container cluster of a cloud platform belongs, allowing different teams to manage their own container clusters. + - Administrators are not visible to tenants within the organization. + - When a regular administrator joins a tenant organization, they default to guest status. + - Tenants are not allowed to create organizations. + +# Incompatible Changes + +- AutoTracing + - To reduce resource overhead and avoid misidentification, the agent by default only parses the following application protocols (to enable parsing of other protocols, please configure `l7-protocol-enabled`): + - HTTP, HTTP2/gRPC, MySQL, Redis, Kafka, DNS, TLS. + - Reminder: When using Wasm to parse private protocols, please add Custom to `l7-protocol-enabled`. +- Agent + - The original environment variable `ONLY_WATCH_K8S_RESOURCE` has been replaced with `K8S_WATCH_POLICY`, [documentation](../ce-install/serverless-pod/). +- API + - Profiling API uses Dataframe return format to compress response size and improve API performance, [PR](https://github.com/deepflowio/deepflow/pull/7011), [documentation](../features/continuous-profiling/data/). + +| | #Functions | Response Size (Byte) | Download Time | +| ------ | ---------- | -------------------- | ------------- | +| Before | 450,000 | 21.9M | 6.16s | +| After | 450,000 | 3.07M | 0.78s | \ No newline at end of file diff --git a/translate/translated/10-release-notes/05-ce-6.5-release.md b/translate/translated/10-release-notes/09-ce-6.5-release.md similarity index 100% rename from translate/translated/10-release-notes/05-ce-6.5-release.md rename to translate/translated/10-release-notes/09-ce-6.5-release.md diff --git a/translate/translated/10-release-notes/06-ee-6.5-release.md b/translate/translated/10-release-notes/10-ee-6.5-release.md similarity index 100% rename from translate/translated/10-release-notes/06-ee-6.5-release.md rename to translate/translated/10-release-notes/10-ee-6.5-release.md diff --git a/translate/translated/10-release-notes/07-ce-6.4-release.md b/translate/translated/10-release-notes/11-ce-6.4-release.md similarity index 100% rename from translate/translated/10-release-notes/07-ce-6.4-release.md rename to translate/translated/10-release-notes/11-ce-6.4-release.md diff --git a/translate/translated/10-release-notes/08-ee-6.4-release.md b/translate/translated/10-release-notes/12-ee-6.4-release.md similarity index 100% rename from translate/translated/10-release-notes/08-ee-6.4-release.md rename to translate/translated/10-release-notes/12-ee-6.4-release.md diff --git a/translate/translated/10-release-notes/09-ce-6.3-release.md b/translate/translated/10-release-notes/13-ce-6.3-release.md similarity index 100% rename from translate/translated/10-release-notes/09-ce-6.3-release.md rename to translate/translated/10-release-notes/13-ce-6.3-release.md diff --git a/translate/translated/10-release-notes/10-ee-6.3-release.md b/translate/translated/10-release-notes/14-ee-6.3-release.md similarity index 100% rename from translate/translated/10-release-notes/10-ee-6.3-release.md rename to translate/translated/10-release-notes/14-ee-6.3-release.md diff --git a/translate/translated/10-release-notes/11-ce-6.2-release.md b/translate/translated/10-release-notes/15-ce-6.2-release.md similarity index 100% rename from translate/translated/10-release-notes/11-ce-6.2-release.md rename to translate/translated/10-release-notes/15-ce-6.2-release.md diff --git a/translate/translated/10-release-notes/12-ee-6.2-release.md b/translate/translated/10-release-notes/16-ee-6.2-release.md similarity index 100% rename from translate/translated/10-release-notes/12-ee-6.2-release.md rename to translate/translated/10-release-notes/16-ee-6.2-release.md diff --git a/translate/translated/10-release-notes/13-ce-6.1-release.md b/translate/translated/10-release-notes/17-ce-6.1-release.md similarity index 100% rename from translate/translated/10-release-notes/13-ce-6.1-release.md rename to translate/translated/10-release-notes/17-ce-6.1-release.md diff --git a/translate/translated/10-release-notes/14-ee-6.1-release.md b/translate/translated/10-release-notes/18-ee-6.1-release.md similarity index 100% rename from translate/translated/10-release-notes/14-ee-6.1-release.md rename to translate/translated/10-release-notes/18-ee-6.1-release.md diff --git a/translate/translated/README.md b/translate/translated/README.md index 0c3171fa..89b31eca 100644 --- a/translate/translated/README.md +++ b/translate/translated/README.md @@ -3,6 +3,8 @@ home: true # This file is very important, do not delete +> This document was translated by ChatGPT + # Title and Description heroText: DeepFlow - Achieve Observability Instantly description: Achieve zero-intrusion (Zero Code) and full-stack (Full Stack) observability instantly using eBPF and Wasm technologies, enabling continuous innovation for cloud-native and AI applications.