Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
66 changes: 66 additions & 0 deletions applicationFE/scripts/test-builtin-helm.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
import assert from 'node:assert/strict'
import { readFile } from 'node:fs/promises'
import { ref, computed, watch, nextTick } from 'vue'
import ts from 'typescript'
const source = await readFile(new URL('../src/views/softwareCatalog/components/applicationInstallationForm.vue', import.meta.url), 'utf8')
function between(start, end) {
const first = source.indexOf(start), last = source.indexOf(end, first)
assert.ok(first >= 0 && last > first)
return source.slice(first, last)
}
const code = ts.transpileModule([
between('const isBuiltInPersistentCatalog =', 'const isJupyterObjectStorageCatalog ='),
between('const supportsStorageClassConfig =', 'const objectStorageEndpointPlaceholder ='),
between('function buildK8sAdditionalConfig()', 'function validateStorageClassSelection()'),
'return { isBuiltInPersistentCatalog, supportsStorageClassConfig, storageClassRequired, storageClassErrorMessage, buildK8sAdditionalConfig, applyBuiltInPersistentDefaults };'
].join('\n'), { compilerOptions: { target: ts.ScriptTarget.ES2022 } }).outputText
function harness() {
const state = Object.fromEntries(Object.entries({
selectInfra: 'K8S', selectedCatalogInfo: {}, selectedCatalogChartName: '', isJupyterObjectStorageCatalog: false,
isLokiCatalog: false, ingressData: { ingressEnabled: true }, hpaData: { hpaEnabled: true, hpaMinReplicas: 3 },
workloadRebalancingEnabled: true, selectedStorageClass: 'standard', storageClassList: [{name: 'standard'}],
storageClassLoading: false, storageClassLoadError: false, storageClassFailure: '', notebookStorageGi: 10,
selectedStorageMinimum: 1, modalTitle: 'Application Installation', showObjectStorageConfig: false, objectStorageData: {enabled:false}
}).map(([key,value]) => [key, ref(value)]))
const env = {...state, computed, watch, hasCatalogCapability: () => false,
_: {isEmpty: value => !value?.length}, buildObjectStorageConfig: () => {throw new Error('Unexpected object storage')}}
return {...state,...new Function(...Object.keys(env),code)(...Object.values(env))}
}
let cases=0
for (const app of ['redis','mariadb','postgresql','apache','tomcat']) {
for (const target of ['VM','K8S']) {
const h=harness()
h.selectInfra.value=target
h.selectedCatalogChartName.value=app
h.selectedCatalogInfo.value={helmChart:{chartName:app,repositoryName:'mcmp-builtin',chartRepositoryUrl:'classpath:helm',chartVersion:'0.1.0',packageId:'mcmp-builtin-'+app}}
await nextTick()
const persistent=target==='K8S' && ['redis','mariadb','postgresql'].includes(app)
assert.equal(h.isBuiltInPersistentCatalog.value,persistent)
assert.equal(h.storageClassRequired.value,persistent)
if (persistent) {
assert.equal(h.ingressData.value.ingressEnabled,false)
assert.equal(h.hpaData.value.hpaEnabled,false)
assert.equal(h.workloadRebalancingEnabled.value,false)
assert.deepEqual(h.buildK8sAdditionalConfig(),{storageClass:'standard',storageSize:'10Gi',storageAccessMode:'ReadWriteOnce'})
for (const size of [0,-1,1.5,10000]) {
h.notebookStorageGi.value=size
assert.match(h.storageClassErrorMessage.value,/whole-number/)
cases++
}
h.notebookStorageGi.value=20; h.selectedStorageMinimum.value=20
assert.equal(h.storageClassErrorMessage.value,'')
h.notebookStorageGi.value=10
assert.match(h.storageClassErrorMessage.value,/20/)
h.selectedStorageClass.value=''
assert.match(h.storageClassErrorMessage.value,/StorageClass/)
h.selectedCatalogInfo.value.helmChart.packageId='custom'
assert.equal(h.isBuiltInPersistentCatalog.value,false)
} else {
assert.equal(h.ingressData.value.ingressEnabled,true)
assert.equal(h.hpaData.value.hpaEnabled,true)
assert.equal(h.buildK8sAdditionalConfig(),undefined)
}
cases++
}
}
console.log(`Built-in Helm form passed (${cases} scenarios plus storage and identity checks).`)
3 changes: 2 additions & 1 deletion applicationFE/scripts/test-vm-clustering.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -90,7 +90,8 @@ const submitStart = form.indexOf('const runInstall = async () => {') + 'const ru
const guardEnd = form.indexOf("\n if (modalTitle.value === 'Application Installation' && (specCheckFlag", submitStart)
assert.ok(guardEnd > submitStart)
const submitGuard = new Function('modalTitle', 'selectInfra', 'selectDeploymentType', 'canSelectClustering', 'toast',
'const deploying = { value: false }, deploymentCompleted = { value: false };\n' + transpile(form.slice(submitStart, guardEnd)) + '\nreturn "continue";')
// The separate provider guard is unrelated to clustering eligibility.
'const deploying = { value: false }, deploymentCompleted = { value: false }, jupyterInstallationUnsupported = { value: false };\n' + transpile(form.slice(submitStart, guardEnd)) + '\nreturn "continue";')
const installation = { value: 'Application Installation' }
let errors = 0
assert.equal(submitGuard(installation, { value: 'VM' }, { value: 'Clustering' }, { value: false }, { error: () => errors++ }), undefined)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -549,10 +549,11 @@
<button type="button" class="btn btn-outline-primary" :disabled="storageCreating || !storageCapability.canCreate || !newStorageClassName" @click="createNotebookStorageClass">{{ storageCreating ? 'Creating...' : 'Create NHN StorageClass' }}</button>
</div>
</div>
<div v-if="isJupyterObjectStorageCatalog" class="mt-2">
<label class="form-label">Notebook volume capacity (GiB)</label>
<div v-if="isJupyterObjectStorageCatalog || isBuiltInPersistentCatalog" class="mt-2">
<label class="form-label">{{ isJupyterObjectStorageCatalog ? 'Notebook' : 'Data' }} volume capacity (GiB)</label>
<input type="number" class="form-control" v-model.number="notebookStorageGi" :min="selectedStorageMinimum" step="1">
<p class="text-muted">Minimum {{ selectedStorageMinimum }} GiB for the known disk limits. Access mode: ReadWriteOnce. Provider quotas and disk availability are checked during provisioning.</p>
<p v-if="isBuiltInPersistentCatalog" class="text-muted">Single instance with generated credentials in Secret &lt;release-name&gt;-auth (key: password). Data PVC and credentials are retained after uninstall; remove them separately when no longer needed.</p>
</div>
</div>

Expand All @@ -577,6 +578,7 @@
class="form-check-input"
type="checkbox"
id="hpaEnabled"
:disabled="isBuiltInPersistentCatalog"
v-model="hpaData.hpaEnabled">
<label class="form-check-label" for="hpaEnabled">
Enable HPA (Horizontal Pod Autoscaler)
Expand Down Expand Up @@ -645,6 +647,7 @@
class="form-check-input"
type="checkbox"
id="workloadRebalancingEnabled"
:disabled="isBuiltInPersistentCatalog"
v-model="workloadRebalancingEnabled">
<label class="form-check-label" for="workloadRebalancingEnabled">
Enable Workload Rebalancing
Expand All @@ -662,10 +665,12 @@
class="form-check-input"
type="checkbox"
id="ingressEnabled"
:disabled="isBuiltInPersistentCatalog"
v-model="ingressData.ingressEnabled">
<label class="form-check-label" for="ingressEnabled">
Enable Ingress
</label>
<p v-if="isBuiltInPersistentCatalog" class="text-muted">This TCP application uses an internal ClusterIP Service, not HTTP Ingress. Use an authenticated port-forward for access from your PC.</p>
</div>
</div>

Expand Down Expand Up @@ -2192,6 +2197,22 @@ const STORAGE_CLASS_CAPABILITY = 'storage-class'
const CONFIG_CAPABILITY_REF_TYPES = ['CAPABILITY', 'TAG']

const isLokiCatalog = computed(() => selectedCatalogChartName.value === 'loki')
const isBuiltInPersistentCatalog = computed(() => {
const chart = selectedCatalogInfo.value?.helmChart
return selectInfra.value === 'K8S' && chart?.repositoryName === 'mcmp-builtin'
&& chart?.chartRepositoryUrl === 'classpath:helm' && chart?.chartVersion === '0.1.0'
&& chart?.packageId === 'mcmp-builtin-' + selectedCatalogChartName.value
&& ['redis', 'mariadb', 'postgresql'].includes(selectedCatalogChartName.value)
})
function applyBuiltInPersistentDefaults() {
if (!isBuiltInPersistentCatalog.value) return
ingressData.value.ingressEnabled = false
hpaData.value.hpaEnabled = false
hpaData.value.hpaMinReplicas = 1
hpaData.value.hpaMaxReplicas = 1
workloadRebalancingEnabled.value = false
}
watch(isBuiltInPersistentCatalog, applyBuiltInPersistentDefaults)
const isJupyterObjectStorageCatalog = computed(() => {
const packageName = String(selectedCatalogInfo.value?.packageInfo?.packageName || '').toLowerCase()
return packageName.includes('jupyter') && hasObjectStorageCapability(selectedCatalogInfo.value as SoftwareCatalog)
Expand All @@ -2208,13 +2229,14 @@ const jupyterInstallationUnsupported = computed(() => {
const supportsStorageClassConfig = computed(() => {
if (selectInfra.value !== 'K8S') return false
if (isJupyterObjectStorageCatalog.value) return true
if (isBuiltInPersistentCatalog.value) return true
if (!selectedCatalogInfo.value?.helmChart) return false
if (!isLokiCatalog.value) return false
return hasCatalogCapability(selectedCatalogInfo.value, STORAGE_CLASS_CAPABILITY)
})

const storageClassRequired = computed(() => {
return supportsStorageClassConfig.value && (isLokiCatalog.value || isJupyterObjectStorageCatalog.value)
return supportsStorageClassConfig.value && (isLokiCatalog.value || isJupyterObjectStorageCatalog.value || isBuiltInPersistentCatalog.value)
})

const showStorageClassConfig = computed(() => {
Expand Down Expand Up @@ -2242,6 +2264,8 @@ const storageClassErrorMessage = computed(() => {
if (_.isEmpty(selectedStorageClass.value)) return 'This application requires a StorageClass.'
if (isJupyterObjectStorageCatalog.value && (!Number.isInteger(notebookStorageGi.value) || notebookStorageGi.value < selectedStorageMinimum.value))
return 'Enter a whole-number notebook capacity of at least ' + selectedStorageMinimum.value + ' GiB.'
if (isBuiltInPersistentCatalog.value && (!Number.isInteger(notebookStorageGi.value) || notebookStorageGi.value < selectedStorageMinimum.value || notebookStorageGi.value > 9999))
return 'Enter a whole-number volume capacity between ' + selectedStorageMinimum.value + ' and 9999 GiB.'
return ''
})

Expand Down Expand Up @@ -2369,7 +2393,7 @@ function buildK8sAdditionalConfig() {
const config = {} as Record<string, any>
if (storageClassRequired.value && !_.isEmpty(selectedStorageClass.value)) {
config.storageClass = selectedStorageClass.value
if (isJupyterObjectStorageCatalog.value) {
if (isJupyterObjectStorageCatalog.value || isBuiltInPersistentCatalog.value) {
config.storageSize = notebookStorageGi.value + 'Gi'
config.storageAccessMode = 'ReadWriteOnce'
}
Expand Down Expand Up @@ -2463,6 +2487,7 @@ const onChangeCatalog = async () => {
ingressTlsEnabled: Boolean(catalogInfo.ingressTlsEnabled),
ingressTlsSecret: catalogInfo.ingressTlsSecret || ''
}
applyBuiltInPersistentDefaults()
if (selectInfra.value === 'K8S' && isJupyterObjectStorageCatalog.value) {
ingressData.value.ingressEnabled = true
hpaData.value.hpaEnabled = false
Expand Down
48 changes: 48 additions & 0 deletions scripts/archive/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,48 @@
# Rebuild daily archive summaries

The native-query aliases must match `DailyAggregationProjection` properties exactly.
With Spring Data JPA 3.2, snake_case aliases could return a null `sampleCount` and
cause every scheduled daily aggregation to be skipped even while raw metrics accumulated.

Fixing the query repairs future runs; it does not automatically fill historical gaps.
`backfill-daily-metrics.sql` rebuilds only selected deployments and completed days
from **real retained measurements**. It never synthesizes measurements or deletes raw data.
It uses the current scheduler's formulas, including 10 minutes per sample and its
inclusive 23:59:59 cutoff. Days without raw samples are not invented or overwritten.

1. Back up the database (`pg_dump -Fc`) and verify the backup's contents.
2. Use the same timezone as the AM scheduler. Review IDs and the retained date range.
3. Preview with an authenticated local PostgreSQL connection:

```sh
psql -X -v deployment_ids=7,11,22,24 \
-v start_date=2026-09-15 -v end_date=2026-09-20 \
-f scripts/archive/backfill-daily-metrics.sql
```

4. Repeat with `-v apply=true` to commit. The default is a rollback. Repeating the
committed operation updates the same summaries rather than creating duplicates.
A rollback may still advance PostgreSQL sequence values; gaps in IDs are harmless.
5. Re-run **policy recommendation analysis** for each deployment through the normal
authenticated `POST /api/applications/{deploymentId}/policy-recommendation/analyze`
endpoint. It recomputes the 90-, 30-, and 7-day results. Otherwise the UI may still
show an older stored analysis. Keep existing analysis rows as historical records.
6. Check `daily_metrics_summary`, the operation-profile API, and the next scheduled
run (02:00 aggregation; 02:10 analysis in the server timezone).

A valid analysis day currently needs at least 72 samples, at least 720 running
minutes, and a CPU or memory P95 value. Fewer than seven valid days still correctly
results in insufficient data after repair.

## Verification

`bash scripts/archive/test-backfill.sh` runs PostgreSQL 16 integration tests in a
disposable Docker container with no external network, published ports, or persistent
data volume. It covers preview rollback, repeatable updates, CPU/memory/network/event
values, null metrics, the 72-sample threshold, date/deployment scoping, source-data
preservation, and invalid-input rejection. The temporary test container is removed
on exit. It requires the `postgres:16-alpine` image to be available or downloadable.

`bash gradlew test --tests '*ResourceMetricsHistoryRepositoryTest'` tests the actual
Spring Data native projection against isolated in-memory data, including every
projection field, empty/single-sample data and 71/72/137/144-sample cases.
99 changes: 99 additions & 0 deletions scripts/archive/backfill-daily-metrics.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,99 @@
-- PostgreSQL / psql only. Back up the database first.
-- Required variables: deployment_ids (comma-separated IDs), start_date, end_date.
-- end_date is inclusive and MUST be before CURRENT_DATE (database/server timezone).
-- Dry-run by default; add -v apply=true only after reviewing the result.
-- Rebuilds derived daily summaries only. Raw metrics, events and logs are untouched.
\set ON_ERROR_STOP on
\if :{?apply}
\else
\set apply false
\endif
BEGIN;
SET LOCAL lock_timeout = '5s';
SET LOCAL statement_timeout = '60s';
CREATE TEMP TABLE archive_backfill_scope ON COMMIT DROP AS
SELECT string_to_array(:'deployment_ids', ',')::bigint[] AS ids,
:'start_date'::date AS first_day, :'end_date'::date AS last_day;
DO $$
BEGIN
IF EXISTS (SELECT 1 FROM archive_backfill_scope
WHERE cardinality(ids) IS NULL OR cardinality(ids) = 0
OR first_day IS NULL OR last_day IS NULL
OR first_day > last_day OR last_day >= CURRENT_DATE) THEN
RAISE EXCEPTION 'Provide deployment IDs and a valid completed-day range';
END IF;
IF EXISTS (SELECT 1 FROM archive_backfill_scope s, unnest(s.ids) AS target(id)
WHERE NOT EXISTS (SELECT 1 FROM deployment_history d WHERE d.id = target.id)) THEN
RAISE EXCEPTION 'Unknown deployment ID';
END IF;
END $$;

WITH raw AS (
SELECT m.deployment_id, m.recorded_at::date AS summary_date,
avg(cpu_usage_pct) AS avg_cpu_pct, max(cpu_usage_pct) AS max_cpu_pct,
percentile_cont(0.95) WITHIN GROUP (ORDER BY cpu_usage_pct) AS p95_cpu_pct,
stddev(cpu_usage_pct) AS stddev_cpu,
avg(memory_usage_pct) AS avg_memory_pct, max(memory_usage_pct) AS max_memory_pct,
percentile_cont(0.95) WITHIN GROUP (ORDER BY memory_usage_pct) AS p95_memory_pct,
stddev(memory_usage_pct) AS stddev_memory,
avg(network_in_bytes) AS avg_network_in_bytes, max(network_in_bytes) AS max_network_in_bytes,
avg(network_out_bytes) AS avg_network_out_bytes, max(network_out_bytes) AS max_network_out_bytes,
count(*)::integer AS sample_count,
(count(*) FILTER (WHERE oom_killed = true))::integer AS oom_count,
(count(*) FILTER (WHERE status = 'RUNNING') * 10)::integer AS running_minutes,
(count(*) * 10)::integer AS total_minutes, min(resource_type) AS resource_type
FROM resource_metrics_history m CROSS JOIN archive_backfill_scope s
WHERE m.deployment_id = ANY(s.ids)
AND m.recorded_at >= s.first_day AND m.recorded_at < s.last_day + 1
-- Match the current scheduler's inclusive 23:59:59 upper bound exactly.
AND m.recorded_at <= m.recorded_at::date + time '23:59:59'
GROUP BY m.deployment_id, m.recorded_at::date
), events AS (
SELECT e.deployment_id, e.occurred_at::date AS summary_date,
(count(*) FILTER (WHERE event_type = 'OOM_KILLED'))::integer AS oom_count,
(count(*) FILTER (WHERE event_type = 'RESTART'))::integer AS restart_count,
(count(*) FILTER (WHERE event_type = 'CRASH_LOOP'))::integer AS crash_loop_count
FROM abnormal_event e CROSS JOIN archive_backfill_scope s
WHERE e.deployment_id = ANY(s.ids)
AND e.occurred_at >= s.first_day AND e.occurred_at < s.last_day + 1
AND e.occurred_at <= e.occurred_at::date + time '23:59:59'
GROUP BY e.deployment_id, e.occurred_at::date
)
INSERT INTO daily_metrics_summary (
deployment_id, summary_date, avg_cpu_pct, max_cpu_pct, p95_cpu_pct, stddev_cpu,
avg_memory_pct, max_memory_pct, p95_memory_pct, stddev_memory,
avg_network_in_bytes, max_network_in_bytes, avg_network_out_bytes, max_network_out_bytes,
sample_count, oom_count, restart_count, crash_loop_count, running_minutes, total_minutes,
resource_type, created_at
)
SELECT r.deployment_id, r.summary_date, avg_cpu_pct, max_cpu_pct, p95_cpu_pct, stddev_cpu,
avg_memory_pct, max_memory_pct, p95_memory_pct, stddev_memory,
avg_network_in_bytes, max_network_in_bytes, avg_network_out_bytes, max_network_out_bytes,
sample_count, r.oom_count + coalesce(e.oom_count, 0), coalesce(e.restart_count, 0),
coalesce(e.crash_loop_count, 0), running_minutes, total_minutes, resource_type, localtimestamp
FROM raw r LEFT JOIN events e USING (deployment_id, summary_date)
ON CONFLICT (deployment_id, summary_date) DO UPDATE SET
avg_cpu_pct = excluded.avg_cpu_pct, max_cpu_pct = excluded.max_cpu_pct,
p95_cpu_pct = excluded.p95_cpu_pct, stddev_cpu = excluded.stddev_cpu,
avg_memory_pct = excluded.avg_memory_pct, max_memory_pct = excluded.max_memory_pct,
p95_memory_pct = excluded.p95_memory_pct, stddev_memory = excluded.stddev_memory,
avg_network_in_bytes = excluded.avg_network_in_bytes, max_network_in_bytes = excluded.max_network_in_bytes,
avg_network_out_bytes = excluded.avg_network_out_bytes, max_network_out_bytes = excluded.max_network_out_bytes,
sample_count = excluded.sample_count, oom_count = excluded.oom_count,
restart_count = excluded.restart_count, crash_loop_count = excluded.crash_loop_count,
running_minutes = excluded.running_minutes, total_minutes = excluded.total_minutes,
resource_type = excluded.resource_type;

SELECT d.deployment_id, min(summary_date) AS first_day, max(summary_date) AS last_day,
count(*) AS summary_days, sum(sample_count) AS samples,
count(*) FILTER (WHERE sample_count >= 72 AND running_minutes >= 720
AND (p95_cpu_pct IS NOT NULL OR p95_memory_pct IS NOT NULL)) AS valid_days
FROM daily_metrics_summary d CROSS JOIN archive_backfill_scope s
WHERE d.deployment_id = ANY(s.ids) AND summary_date BETWEEN s.first_day AND s.last_day
GROUP BY d.deployment_id ORDER BY d.deployment_id;
\if :apply
COMMIT;
\else
ROLLBACK;
\echo 'DRY RUN ONLY: summaries rolled back. Sequences may advance; no source records changed.'
\endif
Loading
Loading