[{"data":1,"prerenderedAt":929},["ShallowReactive",2],{"navigation_docs_en":3,"blog_en_application-health-inspection":289},[4,18,51,254,267,276],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":6},"Getting Started",false,"/docs/getting-started","docs/1.getting-started",[10,14],{"title":11,"path":12,"stem":13},"Introduction","/docs/getting-started/introduction","docs/1.getting-started/1.introduction",{"title":15,"path":16,"stem":17},"Quick Start","/docs/getting-started/quick-start","docs/1.getting-started/2.quick-start",{"title":19,"icon":6,"path":20,"stem":21,"children":22,"page":6},"Features","/docs/features","docs/2.features",[23,27,31,35,39,43,47],{"title":24,"path":25,"stem":26},"Alert Triage","/docs/features/alert-triage","docs/2.features/2.alert-triage",{"title":28,"path":29,"stem":30},"Incident Investigation","/docs/features/incident-investigation","docs/2.features/3.incident-investigation",{"title":32,"path":33,"stem":34},"Deployment Verification","/docs/features/deployment-verification","docs/2.features/4.deployment-verification",{"title":36,"path":37,"stem":38},"Data Exploration","/docs/features/data-exploration","docs/2.features/5.data-exploration",{"title":40,"path":41,"stem":42},"Knowledges","/docs/features/knowledges","docs/2.features/6.knowledges",{"title":44,"path":45,"stem":46},"Castrel Proxy","/docs/features/castrel-proxy","docs/2.features/7.castrel-proxy",{"title":48,"path":49,"stem":50},"Automations","/docs/features/automations","docs/2.features/8.automations",{"title":52,"icon":6,"path":53,"stem":54,"children":55},"Integrations","/docs/integrations","docs/3.integrations/index",[56,57,62,67,72,77,81,85,89,94,99,104,109,113,117,122,127,131,136,141,146,151,156,160,165,170,174,178,183,188,193,198,203,208,212,216,220,224,229,234,239,244,249],{"title":52,"path":53,"stem":54},{"title":58,"path":59,"stem":60,"icon":61},"Prometheus","/docs/integrations/prometheus","docs/3.integrations/1.prometheus","i-simple-icons-prometheus",{"title":63,"path":64,"stem":65,"icon":66},"AWS","/docs/integrations/aws","docs/3.integrations/10.aws","i-simple-icons-amazonwebservices",{"title":68,"path":69,"stem":70,"icon":71},"Alibaba Cloud (Aliyun)","/docs/integrations/aliyun","docs/3.integrations/12.aliyun","i-simple-icons-alibabacloud",{"title":73,"path":74,"stem":75,"icon":76},"Tencent Cloud","/docs/integrations/tencent-cloud","docs/3.integrations/13.tencent-cloud","i-lucide-plug",{"title":78,"path":79,"stem":80,"icon":76},"Huawei Cloud","/docs/integrations/huaweicloud","docs/3.integrations/14.huaweicloud",{"title":82,"path":83,"stem":84,"icon":76},"Volcengine","/docs/integrations/volcengine","docs/3.integrations/15.volcengine",{"title":86,"path":87,"stem":88,"icon":76},"QingFanYun (Cloudwise ITSM)","/docs/integrations/qingfanyun","docs/3.integrations/16.qingfanyun",{"title":90,"path":91,"stem":92,"icon":93},"Grafana","/docs/integrations/grafana","docs/3.integrations/17.grafana","i-simple-icons-grafana",{"title":95,"path":96,"stem":97,"icon":98},"VictoriaMetrics","/docs/integrations/victoriametrics","docs/3.integrations/18.victoriametrics","i-simple-icons-victoriametrics",{"title":100,"path":101,"stem":102,"icon":103},"New Relic","/docs/integrations/new-relic","docs/3.integrations/19.new-relic","i-simple-icons-newrelic",{"title":105,"path":106,"stem":107,"icon":108},"Elasticsearch","/docs/integrations/elasticsearch","docs/3.integrations/2.elasticsearch","i-simple-icons-elasticsearch",{"title":110,"path":111,"stem":112,"icon":76},"Zabbix","/docs/integrations/zabbix","docs/3.integrations/20.zabbix",{"title":114,"path":115,"stem":116,"icon":76},"JianKongBao","/docs/integrations/jiankongbao","docs/3.integrations/21.jiankongbao",{"title":118,"path":119,"stem":120,"icon":121},"PagerDuty","/docs/integrations/pagerduty","docs/3.integrations/22.pagerduty","i-simple-icons-pagerduty",{"title":123,"path":124,"stem":125,"icon":126},"Sentry","/docs/integrations/sentry","docs/3.integrations/23.sentry","i-simple-icons-sentry",{"title":128,"path":129,"stem":130,"icon":76},"Freshworks / Freshservice","/docs/integrations/freshworks","docs/3.integrations/24.freshworks",{"title":132,"path":133,"stem":134,"icon":135},"Linear","/docs/integrations/linear","docs/3.integrations/25.linear","i-simple-icons-linear",{"title":137,"path":138,"stem":139,"icon":140},"ClickHouse","/docs/integrations/clickhouse","docs/3.integrations/26.clickhouse","i-simple-icons-clickhouse",{"title":142,"path":143,"stem":144,"icon":145},"Kubernetes","/docs/integrations/kubernetes","docs/3.integrations/27.kubernetes","i-simple-icons-kubernetes",{"title":147,"path":148,"stem":149,"icon":150},"Terraform Cloud / HCP Terraform","/docs/integrations/terraform","docs/3.integrations/28.terraform","i-simple-icons-terraform",{"title":152,"path":153,"stem":154,"icon":155},"Jenkins","/docs/integrations/jenkins","docs/3.integrations/29.jenkins","i-simple-icons-jenkins",{"title":157,"path":158,"stem":159,"icon":93},"Grafana Loki","/docs/integrations/grafana-loki","docs/3.integrations/3.grafana-loki",{"title":161,"path":162,"stem":163,"icon":164},"Ansible / AWX","/docs/integrations/ansible","docs/3.integrations/30.ansible","i-simple-icons-ansible",{"title":166,"path":167,"stem":168,"icon":169},"GitLab","/docs/integrations/gitlab","docs/3.integrations/31.gitlab","i-simple-icons-gitlab",{"title":171,"path":172,"stem":173,"icon":76},"DingTalk","/docs/integrations/dingtalk","docs/3.integrations/32.dingtalk",{"title":175,"path":176,"stem":177,"icon":76},"Feishu / Lark","/docs/integrations/feishu","docs/3.integrations/33.feishu",{"title":179,"path":180,"stem":181,"icon":182},"Telegram","/docs/integrations/telegram","docs/3.integrations/34.telegram","i-simple-icons-telegram",{"title":184,"path":185,"stem":186,"icon":187},"Email","/docs/integrations/email","docs/3.integrations/35.email","i-simple-icons-gmail",{"title":189,"path":190,"stem":191,"icon":192},"WeiXin Clawbot (Enterprise WeChat)","/docs/integrations/weixin-clawbot","docs/3.integrations/36.weixin-clawbot","i-simple-icons-wechat",{"title":194,"path":195,"stem":196,"icon":197},"Notion","/docs/integrations/notion","docs/3.integrations/37.notion","i-simple-icons-notion",{"title":199,"path":200,"stem":201,"icon":202},"Confluence","/docs/integrations/confluence","docs/3.integrations/38.confluence","i-simple-icons-confluence",{"title":204,"path":205,"stem":206,"icon":207},"Google Docs","/docs/integrations/google-docs","docs/3.integrations/39.google-docs","i-simple-icons-googledocs",{"title":209,"path":210,"stem":211,"icon":93},"Grafana Tempo","/docs/integrations/grafana-tempo","docs/3.integrations/4.grafana-tempo",{"title":213,"path":214,"stem":215,"icon":76},"DingTalk Docs","/docs/integrations/dingtalk-docs","docs/3.integrations/40.dingtalk-docs",{"title":217,"path":218,"stem":219,"icon":76},"LDAP","/docs/integrations/ldap","docs/3.integrations/41.ldap",{"title":221,"path":222,"stem":223,"icon":76},"Dify","/docs/integrations/dify","docs/3.integrations/42.dify",{"title":225,"path":226,"stem":227,"icon":228},"Custom MCP","/docs/integrations/custom-mcp","docs/3.integrations/43.custom-mcp","i-simple-icons-anthropic",{"title":230,"path":231,"stem":232,"icon":233},"GitHub","/docs/integrations/github","docs/3.integrations/5.github","i-simple-icons-github",{"title":235,"path":236,"stem":237,"icon":238},"Slack","/docs/integrations/slack","docs/3.integrations/6.slack","i-simple-icons-slack",{"title":240,"path":241,"stem":242,"icon":243},"Vercel","/docs/integrations/vercel","docs/3.integrations/7.vercel","i-simple-icons-vercel",{"title":245,"path":246,"stem":247,"icon":248},"Graylog","/docs/integrations/graylog","docs/3.integrations/8.graylog","i-simple-icons-graylog",{"title":250,"path":251,"stem":252,"icon":253},"Datadog","/docs/integrations/datadog","docs/3.integrations/9.datadog","i-simple-icons-datadog",{"title":255,"path":256,"stem":257,"children":258,"page":6},"Open Platform","/docs/open-platform","docs/4.open-platform",[259,263],{"title":260,"path":261,"stem":262},"Authentication","/docs/open-platform/authentication","docs/4.open-platform/1.authentication",{"title":264,"path":265,"stem":266},"Knowledges API","/docs/open-platform/knowledges","docs/4.open-platform/2.knowledges",{"title":268,"path":269,"stem":270,"children":271,"page":6},"More","/docs/more","docs/5.more",[272],{"title":273,"path":274,"stem":275},"Roadmap","/docs/more/roadmap","docs/5.more/1.roadmap",{"title":277,"path":278,"stem":279,"children":280,"page":6},"Security","/docs/security","docs/6.security",[281,285],{"title":282,"path":283,"stem":284},"Privacy Policy","/docs/security/privacy-policy","docs/6.security/1.privacy-policy",{"title":286,"path":287,"stem":288},"Terms of Service","/docs/security/terms-of-service","docs/6.security/2.terms-of-service",{"id":290,"title":291,"body":292,"description":918,"extension":919,"meta":920,"navigation":6,"path":925,"seo":926,"stem":927,"__hash__":928},"blogs_en/blogs/4.application-health-inspection.md","Castrel AI Health Inspection: Find and Close Risk Before Business Impact",{"type":293,"value":294,"toc":907},"minimark",[295,299,302,305,361,369,374,380,434,437,444,448,451,490,493,496,499,506,510,513,516,577,584,590,593,603,606,610,613,656,659,663,666,733,743,746,750,753,813,819,825,829,832,886,889,893,898,904],[296,297,298],"p",{},"When alerts are firing and user requests are already failing, the team needs incident troubleshooting. When the business is still healthy but risk is accumulating, the team needs health inspection.",[296,300,301],{},"That period—while there is still time to act—is where Castrel AI health inspection creates value. Castrel AI does not stop at current status. It follows an inspection SOP to evaluate SLOs, alerts, metrics, logs, traces, Kubernetes events, and dependencies; compares them with previous inspection results; distinguishes sustained deterioration from temporary fluctuation; projects which business journey could be affected; and produces a remediation sequence that can be verified.",[296,303,304],{},"Health inspection and incident troubleshooting both use logs, metrics, and traces, but they answer different questions. Incident troubleshooting identifies the cause of an impact that has already occurred and restores service as quickly as possible. Health inspection organizes historical trends, capacity headroom, and dependency relationships into a testable risk judgment before impact occurs—making clear when the team needs to act and what must be true before the risk can be closed.",[306,307,308,324],"table",{},[309,310,311],"thead",{},[312,313,314,318,321],"tr",{},[315,316,317],"th",{},"Dimension",[315,319,320],{},"Health inspection",[315,322,323],{},"Incident troubleshooting",[325,326,327,339,350],"tbody",{},[312,328,329,333,336],{},[330,331,332],"td",{},"Trigger",[330,334,335],{},"Runs regularly or when an early adverse trend appears",[330,337,338],{},"Starts after an alert fires, an SLO is missed, or users are affected",[312,340,341,344,347],{},[330,342,343],{},"Core question",[330,345,346],{},"Is risk accumulating, how could it propagate, and when is action required?",[330,348,349],{},"What caused the current impact, and how can service be restored quickly?",[312,351,352,355,358],{},[330,353,354],{},"Completion condition",[330,356,357],{},"The root cause or risk is controlled, and a reinspection proves the risk chain is closed",[330,359,360],{},"Service is restored, user impact has stopped, and the incident is mitigated",[296,362,363,364,368],{},"This article uses ",[365,366,367],"code",{},"ShopOne",", a representative e-commerce application with 11 business microservices, to show how Castrel AI can identify and close a capacity risk before business impact. Its core checkout journey connects gateway, order, inventory, catalog, payment, and other services through synchronous HTTP calls.",[370,371,373],"h2",{"id":372},"the-latest-dashboard-looked-healthy-but-castrel-ai-classified-the-application-as-warning","The latest dashboard looked healthy, but Castrel AI classified the application as Warning",[296,375,376,377,379],{},"At the start of the inspection, ",[365,378,367],{}," showed no obvious business problem:",[306,381,382,392],{},[309,383,384],{},[312,385,386,389],{},[315,387,388],{},"Current business signal",[315,390,391],{},"Inspection result",[325,393,394,402,410,418,426],{},[312,395,396,399],{},[330,397,398],{},"Gateway availability",[330,400,401],{},"99.96%",[312,403,404,407],{},[330,405,406],{},"Gateway P95 latency",[330,408,409],{},"168 ms, below the 200 ms target",[312,411,412,415],{},[330,413,414],{},"Gateway error rate",[330,416,417],{},"0.04%",[312,419,420,423],{},[330,421,422],{},"Service targets",[330,424,425],{},"All online",[312,427,428,431],{},[330,429,430],{},"Firing critical alerts",[330,432,433],{},"0",[296,435,436],{},"A review that ended with this snapshot could reasonably declare the application healthy. Castrel AI continued with three additional tasks: it compared previous inspection trends, examined evidence from expensive calls, and determined whether the current resource change could propagate into the checkout journey.",[296,438,439,440,443],{},"Castrel AI classified the application as ",[365,441,442],{},"warning",". The reason was not an active business failure. It was a capacity window that was shrinking quickly.",[370,445,447],{"id":446},"castrel-ai-calculated-the-growth-rate-instead-of-merely-reporting-824-disk-usage","Castrel AI calculated the growth rate instead of merely reporting 82.4% disk usage",[296,449,450],{},"A single disk value does not determine an action. Utilization at 82.4% may be a temporary peak or a stable baseline. Castrel AI aligned the current result with the previous 12 hours of inspection data and found a sustained rise:",[306,452,453,464],{},[309,454,455],{},[312,456,457,460],{},[315,458,459],{},"Time",[315,461,463],{"align":462},"right","Disk utilization",[325,465,466,474,482],{},[312,467,468,471],{},[330,469,470],{},"12 hours earlier",[330,472,473],{"align":462},"74.2%",[312,475,476,479],{},[330,477,478],{},"6 hours earlier",[330,480,481],{"align":462},"78.6%",[312,483,484,487],{},[330,485,486],{},"Current",[330,488,489],{"align":462},"82.4%",[296,491,492],{},"Utilization had increased by an average of approximately 0.68 percentage points per hour. A linear projection at the current rate placed the disk at the 90% high-risk boundary in roughly 11 hours and at exhaustion in approximately 26 hours.",[296,494,495],{},"Castrel AI did not present that estimate as a guaranteed prediction. Traffic, data volume, and cleanup behavior could change the actual timing. The risk window nevertheless answered a more useful question than “what is disk usage now?”: how much time remained for the team to intervene without business impact.",[296,497,498],{},"That changed the priority. The disk had not yet crossed a critical alert threshold, but it could no longer wait for a routine capacity review.",[296,500,501],{},[502,503],"img",{"alt":504,"src":505},"Capacity validation evidence from a simulated failure scenario: abnormal SQL caused MySQL temporary-file writes to surge, pushing the root partition to a high level in the same time window.","/images/blog/4.application-health-inspection/disk-en.png",[370,507,509],{"id":508},"castrel-ai-crossed-metrics-logs-and-traces-to-identify-what-was-driving-growth","Castrel AI crossed metrics, logs, and traces to identify what was driving growth",[296,511,512],{},"Expanding the disk based on a trend alone would only delay recurrence. Castrel AI inspected MySQL, service traces, and connection-pool behavior and found several signals moving together in the same window:",[296,514,515],{},"Health inspection does not stop at finding that disk utilization is elevated. The team needs to establish the source of resource pressure, its propagation path, and its business consequence before deciding whether it is an observable fluctuation or a risk that requires early action.",[306,517,518,531],{},[309,519,520],{},[312,521,522,525,528],{},[315,523,524],{},"Evidence",[315,526,527],{},"Current change",[315,529,530],{},"Castrel AI judgment",[325,532,533,544,555,566],{},[312,534,535,538,541],{},[330,536,537],{},"MySQL query P95",[330,539,540],{},"Increased from about 1.4 seconds to 5.8 seconds",[330,542,543],{},"Database workload was rising rapidly",[312,545,546,549,552],{},[330,547,548],{},"On-disk temporary-table creation rate",[330,550,551],{},"Reached 3.2 times the historical baseline",[330,553,554],{},"More query results were spilling to disk",[312,556,557,560,563],{},[330,558,559],{},"Order connection pool",[330,561,562],{},"7 of 10 connections remained active; waiters increased from 0 to 4",[330,564,565],{},"The pool was not exhausted, but downstream queries were beginning to hold connections",[312,567,568,571,574],{},[330,569,570],{},"Tempo slow calls",[330,572,573],{},"Inventory and catalog slow calls pointed to the same query pattern",[330,575,576],{},"Risk was crossing services in checkout dependencies",[296,578,579,580,583],{},"The SQL captured in the traces included ",[365,581,582],{},"JOIN user_behavior_log ubl ON TRUE",". Without an effective join condition, the query produced a Cartesian-product pattern. As the result set grew, MySQL wrote more temporary data and held database connections for longer.",[296,585,586],{},[502,587],{"alt":588,"src":589},"Root-cause validation evidence from a simulated failure scenario: abnormal SQL, MySQL temporary-file write-failure logs, and the investigation conclusion establish the source of resource pressure.","/images/blog/4.application-health-inspection/sql-en.png",[296,591,592],{},"Castrel AI therefore did not report four unrelated anomalies. It constructed a testable propagation path:",[594,595,601],"pre",{"className":596,"code":598,"language":599,"meta":600},[597],"language-text","Abnormal SQL multiplies the result set\n    → on-disk temporary tables and files continue to grow\n    → free disk space falls rapidly\n    → slow queries hold database connections for longer\n    → the order connection pool queues and times out\n    → checkout latency and failures rise\n","text","",[365,602,598],{"__ignoreMap":600},[296,604,605],{},"The last two stages had not yet occurred. That is precisely why the chain was actionable. A current-state dashboard could describe each component as it was; Castrel AI combined trend, dependency, and historical evidence to determine how the resource pressure could turn into business risk.",[370,607,609],{"id":608},"castrel-ai-turned-the-risk-judgment-into-an-ordered-remediation-plan","Castrel AI turned the risk judgment into an ordered remediation plan",[296,611,612],{},"Castrel AI did not treat “add disk” as the complete answer. It ordered actions according to the failure chain:",[614,615,616,632,638,644,650],"ol",{},[617,618,619,623,624,627,628,631],"li",{},[620,621,622],"strong",{},"Stop the source of growth:"," inspect the ",[365,625,626],{},"user_behavior_log"," query generator, remove the ",[365,629,630],{},"ON TRUE"," Cartesian join, and add the intended business join condition.",[617,633,634,637],{},[620,635,636],{},"Restore safe headroom:"," remove temporary files generated by the abnormal query and return disk utilization to a safe range.",[617,639,640,643],{},[620,641,642],{},"Reduce recurrence risk:"," optimize large-result sorting and deep pagination, then add dedicated capacity and growth-rate alerts for temporary-file storage.",[617,645,646,649],{},[620,647,648],{},"Watch the amplifier:"," monitor active pool connections, waiters, and timeouts without using a larger pool as a substitute for the SQL fix.",[617,651,652,655],{},[620,653,654],{},"Run the same inspection again:"," verify the result with the same SLOs, windows, and risk chain.",[296,657,658],{},"This sequence distinguishes Castrel AI from a fixed-threshold script. A script can notify an operator when disk usage reaches a configured percentage. Castrel AI also explains what is driving growth, which business journey may be affected, what to fix first, and how to verify that the problem has not merely disappeared temporarily.",[370,660,662],{"id":661},"the-second-inspection-verified-risk-closurenot-merely-the-absence-of-alerts","The second inspection verified risk closure—not merely the absence of alerts",[296,664,665],{},"After the team corrected the join condition and removed the temporary files, it ran the same Castrel AI inspection task again. Castrel AI reused the first report’s windows, metric definitions, and risk chain to verify each result:",[306,667,668,681],{},[309,669,670],{},[312,671,672,675,678],{},[315,673,674],{},"Verification item",[315,676,677],{"align":462},"Before remediation",[315,679,680],{"align":462},"Reinspection result",[325,682,683,693,703,714,724],{},[312,684,685,687,690],{},[330,686,463],{},[330,688,689],{"align":462},"82.4%, growing about 0.68 percentage points per hour",[330,691,692],{"align":462},"63.1%, remaining between 63.0% and 63.4% over the next 6 hours",[312,694,695,697,700],{},[330,696,537],{},[330,698,699],{"align":462},"5.8 seconds",[330,701,702],{"align":462},"220 ms",[312,704,705,708,711],{},[330,706,707],{},"On-disk temporary-table rate",[330,709,710],{"align":462},"3.2 times the historical baseline",[330,712,713],{"align":462},"Returned close to baseline",[312,715,716,718,721],{},[330,717,559],{},[330,719,720],{"align":462},"7/10 active with 4 waiters",[330,722,723],{"align":462},"3–4/10 active with no waiters",[312,725,726,728,730],{},[330,727,398],{},[330,729,401],{"align":462},[330,731,732],{"align":462},"99.98%",[296,734,735,736,738,739,742],{},"Castrel AI changed the state from ",[365,737,442],{}," to ",[365,740,741],{},"healthy",", but not because the number of firing critical alerts remained zero. Closure required the disk trend to flatten, abnormal query latency to recover, temporary-file pressure to fall, pool waiters to disappear, and core business signals to remain healthy.",[296,744,745],{},"The first inspection created an intervention window. The second produced reviewable evidence that the risk was closed. Health inspection became a continuous process from detection and judgment to action and verification—not a recurring report that no one follows up.",[370,747,749],{"id":748},"a-health-inspection-delivers-an-executable-risk-loopnot-an-anomaly-list","A health inspection delivers an executable risk loop—not an anomaly list",[296,751,752],{},"Each Castrel AI health inspection preserves its reasoning and organizes it into a result that teams can hand off, execute, and validate again in the next inspection:",[306,754,755,765],{},[309,756,757],{},[312,758,759,762],{},[315,760,761],{},"Deliverable",[315,763,764],{},"What the team can do with it",[325,766,767,781,789,797,805],{},[312,768,769,772],{},[330,770,771],{},"Health state and business judgment",[330,773,774,775,777,778,780],{},"Determine whether the application is ",[365,776,741],{},", ",[365,779,442],{},", or requires escalation, and whether checkout is affected",[312,782,783,786],{},[330,784,785],{},"Risk window and priority",[330,787,788],{},"Decide when a resource may cross its risk threshold and whether action belongs in the current shift, that day, or a later iteration",[312,790,791,794],{},[330,792,793],{},"Evidence chain and root-cause judgment",[330,795,796],{},"Understand why the issue can affect the business through evidence from trends, SQL, logs, traces, and connection pools",[312,798,799,802],{},[330,800,801],{},"Ordered remediation plan",[330,803,804],{},"Distinguish what to do first, which actions only mitigate the issue, and which measures prevent recurrence",[312,806,807,810],{},[330,808,809],{},"Reinspection criteria and historical baseline",[330,811,812],{},"Define when the risk is actually closed and let the next inspection validate this conclusion against the current baseline",[296,814,815,816,818],{},"In the ",[365,817,367],{}," scenario, the team receives more than the monitoring observation that disk utilization is 82.4%. It receives an executable risk loop: checkout is still healthy, but capacity headroom is shrinking; abnormal SQL is driving growth; connection-pool pressure could carry the database problem into ordering; the SQL fix takes priority over disk expansion; and the risk can close only when disk trend, query latency, temporary files, connection pools, and business signals recover together.",[296,820,821],{},[502,822],{"alt":823,"src":824},"Castrel AI remediation and reinspection plan: fix the abnormal SQL first, restore capacity headroom, protect the critical path, and verify risk closure against explicit conditions.","/images/blog/4.application-health-inspection/dispositions-en.png",[370,826,828],{"id":827},"castrel-ai-preserves-a-baseline-for-each-operational-domain-then-connects-them-at-the-business-journey","Castrel AI preserves a baseline for each operational domain, then connects them at the business journey",[296,830,831],{},"Capacity risk is easy to miss because each operational domain can appear to be below its own threshold when viewed in isolation. Applications, services, hosts, infrastructure, and MySQL face different risks and should not rely on one generic threshold set. Castrel AI can maintain an independent task and historical baseline for each domain, then relate their impact in an application inspection:",[306,833,834,844],{},[309,835,836],{},[312,837,838,841],{},[315,839,840],{},"Inspection target",[315,842,843],{},"What Castrel AI needs to determine",[325,845,846,854,862,870,878],{},[312,847,848,851],{},[330,849,850],{},"Application",[330,852,853],{},"Whether a core business journey is accumulating cross-service risk",[312,855,856,859],{},[330,857,858],{},"Service",[330,860,861],{},"Whether errors, latency, restarts, and resource pressure are worsening",[312,863,864,867],{},[330,865,866],{},"Host",[330,868,869],{},"How long CPU, memory, disk, and network capacity can support the workload",[312,871,872,875],{},[330,873,874],{},"Infrastructure",[330,876,877],{},"Whether cluster events and saturation are affecting more workloads",[312,879,880,883],{},[330,881,882],{},"MySQL",[330,884,885],{},"Whether slow queries, connections, temporary tables, and disk behavior could amplify into application failure",[296,887,888],{},"The available checks depend on connected integrations, permissions, and inspection rules. The Castrel AI decision model remains consistent: read current state, compare historical trend, relate upstream and downstream evidence, project business consequences, prioritize remediation, and verify closure through reinspection.",[370,890,892],{"id":891},"health-inspection-produces-more-than-a-report-an-earlier-risk-decision","Health inspection produces more than a report: an earlier risk decision",[296,894,815,895,897],{},[365,896,367],{}," scenario, Castrel AI inspected an application with healthy business signals and no critical alerts. It did not repeat the dashboard conclusion. It performed the work that a dashboard alone could not complete:",[594,899,902],{"className":900,"code":901,"language":599,"meta":600},[597],"Verify that the business is healthy now\n    → calculate the disk growth rate and risk window\n    → relate abnormal SQL, temporary files, and connection pressure\n    → project the propagation path into checkout\n    → produce a root-cause-ordered remediation plan\n    → reinspect after remediation and verify closure\n",[365,903,901],{"__ignoreMap":600},[296,905,906],{},"That is the core value of Castrel AI health inspection: it does not wait for failure and explain the past. It identifies future risk while the business is still healthy and converts fragmented evidence into an operational decision that can be executed, reviewed, and reused. It preserves the reasoning, defines remediation order and closure conditions, and becomes the historical baseline for the next inspection.",{"title":600,"searchDepth":908,"depth":908,"links":909},2,[910,911,912,913,914,915,916,917],{"id":372,"depth":908,"text":373},{"id":446,"depth":908,"text":447},{"id":508,"depth":908,"text":509},{"id":608,"depth":908,"text":609},{"id":661,"depth":908,"text":662},{"id":748,"depth":908,"text":749},{"id":827,"depth":908,"text":828},{"id":891,"depth":908,"text":892},"Castrel AI does not wait for incident alerts. It correlates historical trends, SLOs, metrics, logs, and traces to calculate a risk window, identify the abnormal path increasing resource pressure, and verify closure with a post-remediation inspection.","md",{"date":921,"order":922,"category":923,"image":924},"2026-08-17",4,"Product",{"src":505},"/blogs/application-health-inspection",{"ogImage":505,"title":291,"description":918},"blogs/4.application-health-inspection","bFgmur1G1x6xf2WTmBFQcKchhn1rKU--gFEoVzJSJPM",1787301618809]