[{"data":1,"prerenderedAt":519},["ShallowReactive",2],{"navigation_docs_en":3,"blog_en_parallel-incident-investigations":289},[4,18,51,254,267,276],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":6},"Getting Started",false,"/docs/getting-started","docs/1.getting-started",[10,14],{"title":11,"path":12,"stem":13},"Introduction","/docs/getting-started/introduction","docs/1.getting-started/1.introduction",{"title":15,"path":16,"stem":17},"Quick Start","/docs/getting-started/quick-start","docs/1.getting-started/2.quick-start",{"title":19,"icon":6,"path":20,"stem":21,"children":22,"page":6},"Features","/docs/features","docs/2.features",[23,27,31,35,39,43,47],{"title":24,"path":25,"stem":26},"Alert Triage","/docs/features/alert-triage","docs/2.features/2.alert-triage",{"title":28,"path":29,"stem":30},"Incident Investigation","/docs/features/incident-investigation","docs/2.features/3.incident-investigation",{"title":32,"path":33,"stem":34},"Deployment Verification","/docs/features/deployment-verification","docs/2.features/4.deployment-verification",{"title":36,"path":37,"stem":38},"Data Exploration","/docs/features/data-exploration","docs/2.features/5.data-exploration",{"title":40,"path":41,"stem":42},"Knowledges","/docs/features/knowledges","docs/2.features/6.knowledges",{"title":44,"path":45,"stem":46},"Castrel Proxy","/docs/features/castrel-proxy","docs/2.features/7.castrel-proxy",{"title":48,"path":49,"stem":50},"Automations","/docs/features/automations","docs/2.features/8.automations",{"title":52,"icon":6,"path":53,"stem":54,"children":55},"Integrations","/docs/integrations","docs/3.integrations/index",[56,57,62,67,72,77,81,85,89,94,99,104,109,113,117,122,127,131,136,141,146,151,156,160,165,170,174,178,183,188,193,198,203,208,212,216,220,224,229,234,239,244,249],{"title":52,"path":53,"stem":54},{"title":58,"path":59,"stem":60,"icon":61},"Prometheus","/docs/integrations/prometheus","docs/3.integrations/1.prometheus","i-simple-icons-prometheus",{"title":63,"path":64,"stem":65,"icon":66},"AWS","/docs/integrations/aws","docs/3.integrations/10.aws","i-simple-icons-amazonwebservices",{"title":68,"path":69,"stem":70,"icon":71},"Alibaba Cloud (Aliyun)","/docs/integrations/aliyun","docs/3.integrations/12.aliyun","i-simple-icons-alibabacloud",{"title":73,"path":74,"stem":75,"icon":76},"Tencent Cloud","/docs/integrations/tencent-cloud","docs/3.integrations/13.tencent-cloud","i-lucide-plug",{"title":78,"path":79,"stem":80,"icon":76},"Huawei Cloud","/docs/integrations/huaweicloud","docs/3.integrations/14.huaweicloud",{"title":82,"path":83,"stem":84,"icon":76},"Volcengine","/docs/integrations/volcengine","docs/3.integrations/15.volcengine",{"title":86,"path":87,"stem":88,"icon":76},"QingFanYun (Cloudwise ITSM)","/docs/integrations/qingfanyun","docs/3.integrations/16.qingfanyun",{"title":90,"path":91,"stem":92,"icon":93},"Grafana","/docs/integrations/grafana","docs/3.integrations/17.grafana","i-simple-icons-grafana",{"title":95,"path":96,"stem":97,"icon":98},"VictoriaMetrics","/docs/integrations/victoriametrics","docs/3.integrations/18.victoriametrics","i-simple-icons-victoriametrics",{"title":100,"path":101,"stem":102,"icon":103},"New Relic","/docs/integrations/new-relic","docs/3.integrations/19.new-relic","i-simple-icons-newrelic",{"title":105,"path":106,"stem":107,"icon":108},"Elasticsearch","/docs/integrations/elasticsearch","docs/3.integrations/2.elasticsearch","i-simple-icons-elasticsearch",{"title":110,"path":111,"stem":112,"icon":76},"Zabbix","/docs/integrations/zabbix","docs/3.integrations/20.zabbix",{"title":114,"path":115,"stem":116,"icon":76},"JianKongBao","/docs/integrations/jiankongbao","docs/3.integrations/21.jiankongbao",{"title":118,"path":119,"stem":120,"icon":121},"PagerDuty","/docs/integrations/pagerduty","docs/3.integrations/22.pagerduty","i-simple-icons-pagerduty",{"title":123,"path":124,"stem":125,"icon":126},"Sentry","/docs/integrations/sentry","docs/3.integrations/23.sentry","i-simple-icons-sentry",{"title":128,"path":129,"stem":130,"icon":76},"Freshworks / Freshservice","/docs/integrations/freshworks","docs/3.integrations/24.freshworks",{"title":132,"path":133,"stem":134,"icon":135},"Linear","/docs/integrations/linear","docs/3.integrations/25.linear","i-simple-icons-linear",{"title":137,"path":138,"stem":139,"icon":140},"ClickHouse","/docs/integrations/clickhouse","docs/3.integrations/26.clickhouse","i-simple-icons-clickhouse",{"title":142,"path":143,"stem":144,"icon":145},"Kubernetes","/docs/integrations/kubernetes","docs/3.integrations/27.kubernetes","i-simple-icons-kubernetes",{"title":147,"path":148,"stem":149,"icon":150},"Terraform Cloud / HCP Terraform","/docs/integrations/terraform","docs/3.integrations/28.terraform","i-simple-icons-terraform",{"title":152,"path":153,"stem":154,"icon":155},"Jenkins","/docs/integrations/jenkins","docs/3.integrations/29.jenkins","i-simple-icons-jenkins",{"title":157,"path":158,"stem":159,"icon":93},"Grafana Loki","/docs/integrations/grafana-loki","docs/3.integrations/3.grafana-loki",{"title":161,"path":162,"stem":163,"icon":164},"Ansible / AWX","/docs/integrations/ansible","docs/3.integrations/30.ansible","i-simple-icons-ansible",{"title":166,"path":167,"stem":168,"icon":169},"GitLab","/docs/integrations/gitlab","docs/3.integrations/31.gitlab","i-simple-icons-gitlab",{"title":171,"path":172,"stem":173,"icon":76},"DingTalk","/docs/integrations/dingtalk","docs/3.integrations/32.dingtalk",{"title":175,"path":176,"stem":177,"icon":76},"Feishu / Lark","/docs/integrations/feishu","docs/3.integrations/33.feishu",{"title":179,"path":180,"stem":181,"icon":182},"Telegram","/docs/integrations/telegram","docs/3.integrations/34.telegram","i-simple-icons-telegram",{"title":184,"path":185,"stem":186,"icon":187},"Email","/docs/integrations/email","docs/3.integrations/35.email","i-simple-icons-gmail",{"title":189,"path":190,"stem":191,"icon":192},"WeiXin Clawbot (Enterprise WeChat)","/docs/integrations/weixin-clawbot","docs/3.integrations/36.weixin-clawbot","i-simple-icons-wechat",{"title":194,"path":195,"stem":196,"icon":197},"Notion","/docs/integrations/notion","docs/3.integrations/37.notion","i-simple-icons-notion",{"title":199,"path":200,"stem":201,"icon":202},"Confluence","/docs/integrations/confluence","docs/3.integrations/38.confluence","i-simple-icons-confluence",{"title":204,"path":205,"stem":206,"icon":207},"Google Docs","/docs/integrations/google-docs","docs/3.integrations/39.google-docs","i-simple-icons-googledocs",{"title":209,"path":210,"stem":211,"icon":93},"Grafana Tempo","/docs/integrations/grafana-tempo","docs/3.integrations/4.grafana-tempo",{"title":213,"path":214,"stem":215,"icon":76},"DingTalk Docs","/docs/integrations/dingtalk-docs","docs/3.integrations/40.dingtalk-docs",{"title":217,"path":218,"stem":219,"icon":76},"LDAP","/docs/integrations/ldap","docs/3.integrations/41.ldap",{"title":221,"path":222,"stem":223,"icon":76},"Dify","/docs/integrations/dify","docs/3.integrations/42.dify",{"title":225,"path":226,"stem":227,"icon":228},"Custom MCP","/docs/integrations/custom-mcp","docs/3.integrations/43.custom-mcp","i-simple-icons-anthropic",{"title":230,"path":231,"stem":232,"icon":233},"GitHub","/docs/integrations/github","docs/3.integrations/5.github","i-simple-icons-github",{"title":235,"path":236,"stem":237,"icon":238},"Slack","/docs/integrations/slack","docs/3.integrations/6.slack","i-simple-icons-slack",{"title":240,"path":241,"stem":242,"icon":243},"Vercel","/docs/integrations/vercel","docs/3.integrations/7.vercel","i-simple-icons-vercel",{"title":245,"path":246,"stem":247,"icon":248},"Graylog","/docs/integrations/graylog","docs/3.integrations/8.graylog","i-simple-icons-graylog",{"title":250,"path":251,"stem":252,"icon":253},"Datadog","/docs/integrations/datadog","docs/3.integrations/9.datadog","i-simple-icons-datadog",{"title":255,"path":256,"stem":257,"children":258,"page":6},"Open Platform","/docs/open-platform","docs/4.open-platform",[259,263],{"title":260,"path":261,"stem":262},"Authentication","/docs/open-platform/authentication","docs/4.open-platform/1.authentication",{"title":264,"path":265,"stem":266},"Knowledges API","/docs/open-platform/knowledges","docs/4.open-platform/2.knowledges",{"title":268,"path":269,"stem":270,"children":271,"page":6},"More","/docs/more","docs/5.more",[272],{"title":273,"path":274,"stem":275},"Roadmap","/docs/more/roadmap","docs/5.more/1.roadmap",{"title":277,"path":278,"stem":279,"children":280,"page":6},"Security","/docs/security","docs/6.security",[281,285],{"title":282,"path":283,"stem":284},"Privacy Policy","/docs/security/privacy-policy","docs/6.security/1.privacy-policy",{"title":286,"path":287,"stem":288},"Terms of Service","/docs/security/terms-of-service","docs/6.security/2.terms-of-service",{"id":290,"title":291,"body":292,"description":507,"extension":508,"meta":509,"navigation":6,"path":515,"seo":516,"stem":517,"__hash__":518},"blogs_en/blogs/5.parallel-incident-investigations.md","How Castrel AI Helps Teams Investigate One Incident in Parallel",{"type":293,"value":294,"toc":499},"minimark",[295,299,302,305,310,322,325,406,409,416,420,430,433,436,442,446,449,452,463,469,473,476,487,490,496],[296,297,298],"p",{},"Complex incidents quickly split into several lines of investigation. When database writes stop, one responder needs to explain why an instance cannot start while another checks storage and replica health. Service owners assess impact, and the observability team investigates why monitoring gave no early warning. The questions are related, but placing them in one execution queue slows the response.",[296,300,301],{},"Teams have often divided this work across chat and separate task records. Work begins in parallel, but the incident background, current progress, and supporting evidence end up in different places. The incident lead has to chase updates, and an incoming responder must reconstruct the timeline. When one investigation corrects an early assumption, other responders may continue working from stale information.",[296,303,304],{},"Castrel AI turns the incident detail into a shared workspace for collaborative investigation. One incident can contain multiple investigation tasks running in parallel from the same context. The activity feed and comments preserve the collaboration, while findings and artifacts return to the incident detail. The team can continue asking questions across the incident, have Castrel AI create another investigation, and generate an incident report when the response is ready for review.",[306,307,309],"h2",{"id":308},"four-investigations-move-forward-under-one-incident","Four investigations move forward under one incident",[296,311,312,313,317,318,321],{},"Consider a practical PostgreSQL investigation. A full data volume leaves ",[314,315,316],"code",{},"postgres-cluster-1"," in ",[314,319,320],{},"CrashLoopBackOff",", interrupting database writes in a CloudNativePG cluster. A manual health inspection discovered this incident, and the current record has no related alerts. When alerts are associated with an incident, they are also available as shared context for its investigation tasks.",[296,323,324],{},"At discovery, the team needs to establish the immediate cause, storage state, business impact, and monitoring gap at the same time. Castrel AI supports several ways to start those tasks:",[326,327,328,347],"table",{},[329,330,331],"thead",{},[332,333,334,338,341,344],"tr",{},[335,336,337],"th",{},"Investigation task",[335,339,340],{},"How it starts",[335,342,343],{},"Question to answer",[335,345,346],{},"Result returned to the incident",[348,349,350,365,379,392],"tbody",{},[332,351,352,356,359,362],{},[353,354,355],"td",{},"Investigate the production PostgreSQL disk-full failure",[353,357,358],{},"Started automatically by Castrel AI",[353,360,361],{},"Why can the instance not start, and how can writes be restored?",[353,363,364],{},"Immediate cause, response history, and recovery conclusion",[332,366,367,370,373,376],{},[353,368,369],{},"Check disk status",[353,371,372],{},"Started manually by a responder",[353,374,375],{},"Did volume expansion complete, and are instance and replication states healthy?",[353,377,378],{},"Storage state, role changes, and open validation items",[332,380,381,384,386,389],{},[353,382,383],{},"Assess service interruption and impact",[353,385,372],{},[353,387,388],{},"Which services were affected, and has the business recovered?",[353,390,391],{},"Impact scope and recovery status",[332,393,394,397,400,403],{},[353,395,396],{},"Investigate the missing PostgreSQL disk alert",[353,398,399],{},"Started through the incident assistant",[353,401,402],{},"Why was there no early warning, and which metrics or rules are missing?",[353,404,405],{},"Monitoring gaps with supporting evidence and corrective actions",[296,407,408],{},"The tasks can run together, and responders can add another one when the investigation raises a new question. Each task keeps its launch method, execution state, and full working history while sharing the incident description, affected application, related alerts when present, and current findings. Responders do not need to rewrite the background for every assignment or wait for a single master task to reach their question.",[296,410,411],{},[412,413],"img",{"alt":414,"src":415},"The incident detail lists four investigation tasks with their initiators and completion status.","/images/blog/5.parallel-incident-investigations/parallel-tasks-en.png",[306,417,419],{"id":418},"new-facts-reach-every-workstream-through-the-activity-feed","New facts reach every workstream through the activity feed",[296,421,422,423,425,426,429],{},"Keeping facts current is the hard part of parallel investigation. The original description in this PostgreSQL scenario identified ",[314,424,316],{}," as the primary and suggested that it was the only volume left unexpanded. Later investigation corrected both assumptions: the actual primary was ",[314,427,428],{},"postgres-cluster-2",", and all three data volumes had been expanded to 500 GiB.",[296,431,432],{},"Those corrections change what responders should check next. The Castrel AI activity feed records task status changes and new findings. It also retains comments and the selection of a root-cause report. After storage and role changes were confirmed, a responder used a comment to explain the new primary-replica relationship so the replication work could continue against the right targets. Other responders received the updated basis for their work without waiting for another verbal handoff.",[296,434,435],{},"The incident detail also supports comments and replies. Responders can add a manual confirmation, raise an open question, or describe how one action affects another investigation. The discussion stays attached to the incident, so an incoming responder can see both the conclusion and the conditions under which it was reached.",[296,437,438],{},[412,439],{"alt":440,"src":441},"The incident activity feed records comments, key findings, and task creation, completion, or interruption.","/images/blog/5.parallel-incident-investigations/activity-collaboration-en.png",[306,443,445],{"id":444},"an-incident-question-can-become-a-new-investigation","An incident question can become a new investigation",[296,447,448],{},"The Castrel AI assistant on the incident detail uses the current incident as its context. The incident lead can ask which causes are confirmed, what still blocks recovery, or which risks remain after a role change. A follow-up question can also become an investigation task.",[296,450,451],{},"In this scenario, a responder asked Castrel AI to create a task and investigate why there had been no early warning and which metrics or alert rules were missing. Castrel AI created the task, checked the production Prometheus metrics and rules, then returned the conclusion to the incident as a finding.",[296,453,454,455,458,459,462],{},"The investigation found no usable ",[314,456,457],{},"kubelet_volume_stats_*"," series, no ",[314,460,461],{},"cnpg_*"," metrics, and no disk-capacity or PVC-usage rules among the 156 existing alert rules. The missed warning therefore involved both collection and alerting. The responder did not have to leave the incident, create a separate task, copy the background, and manually move the result back.",[296,464,465],{},[412,466],{"alt":467,"src":468},"The incident assistant creates an investigation from a responder's question and presents evidence about the monitoring gap.","/images/blog/5.parallel-incident-investigations/incident-assistant-en.png",[306,470,472],{"id":471},"task-results-return-to-the-incident-and-form-a-root-cause-report","Task results return to the incident and form a root-cause report",[296,474,475],{},"Parallel investigations produce separate findings and artifacts. The Castrel AI incident detail provides a single task entry point and puts the current status beside two collections: Findings and Artifacts. The incident lead can review conclusions and core evidence at the incident level, opening an individual task only when the complete execution history is needed.",[296,477,478,479,482,483,486],{},"This combined view also separates confirmed facts, corrected assumptions, and open questions. The investigation confirmed that disk exhaustion prevented the instance from starting, all three volumes had been expanded, and database writes had recovered. It also corrected the original primary-role assumption. The deeper reason the disk filled remains open because the required usage telemetry was unavailable; a later check with ",[314,480,481],{},"df",", ",[314,484,485],{},"du",", or restored collection is still needed.",[296,488,489],{},"As the response closes, the team can ask Castrel AI to generate an incident investigation report from the incident context, activity history, task artifacts, and findings. The report preserves the conclusion, timeline, factual corrections, impact, corrective actions, and remaining work. It can then be marked as the incident's root-cause report for post-incident review and handoff.",[296,491,492],{},[412,493],{"alt":494,"src":495},"Castrel AI consolidates the investigations into a report with the incident conclusion and timeline.","/images/blog/5.parallel-incident-investigations/investigation-report-en.png",[296,497,498],{},"After service recovery, the unanswered question does not disappear into someone's chat history. The root-cause report still records that the cause of disk growth requires further verification, giving the next responder a precise place to continue. When a similar incident occurs, the same record provides a reusable task structure, decision evidence, and monitoring follow-up.",{"title":500,"searchDepth":501,"depth":501,"links":502},"",2,[503,504,505,506],{"id":308,"depth":501,"text":309},{"id":418,"depth":501,"text":419},{"id":444,"depth":501,"text":445},{"id":471,"depth":501,"text":472},"Castrel AI lets teams run multiple investigations under one incident while the activity feed and threaded comments keep responders aligned. Findings and artifacts return to the incident detail, where the team can ask follow-up questions, start new investigations, and generate an incident report.","md",{"date":510,"order":511,"category":512,"image":513},"2026-09-07",5,"Product",{"src":514},"/images/blog/5.parallel-incident-investigations/cover-en.png","/blogs/parallel-incident-investigations",{"ogImage":514,"title":291,"description":507},"blogs/5.parallel-incident-investigations","ROSGnA1a1ZF3gWXLV8d3nlgGBvQGxGT440kHzQcAd3M",1789185073475]