From bc943ff652343da63b0f848ad8abb479e7adc5bc Mon Sep 17 00:00:00 2001 From: Uday Raj Sahai Date: Thu, 9 Jul 2026 15:37:03 +0530 Subject: [PATCH 1/4] Unit testing implemented --- .../CompletedTestCases.test.tsx | 122 ++++++++++ .../EvaluateOutputSection.test.ts | 14 ++ .../ManualTestCaseDetailSheet.test.tsx | 72 ++++++ .../RecommendationModal.test.tsx | 73 ++++++ .../manual-evaluation/utils.test.ts | 199 ++++++++++++++++ .../ai-maker/utils/map-audit-results.test.ts | 149 ++++++++++++ .../components/AddIssueModal.test.tsx | 114 +++++++++ .../components/AuditResultsList.test.tsx | 142 +++++++++++ .../BulkTestCaseDetailSheet.test.tsx | 109 +++++++++ .../hooks/use-audit-actions.test.ts | 150 ++++++++++++ .../hooks/use-audit-data.test.ts | 108 +++++++++ .../hooks/use-audit-polling.test.ts | 133 +++++++++++ .../hooks/use-evaluation-detail.test.ts | 133 +++++++++++ features/dashboard/utils/evaluation.test.ts | 67 ++++++ package.json | 3 + .../fixtures/bulk-evaluation/audit-results.ts | 143 +++++++++++ testing/fixtures/bulk-evaluation/audit.ts | 26 ++ testing/fixtures/bulk-evaluation/index.ts | 2 + .../fixtures/playground-evaluation/audit.ts | 27 +++ .../fixtures/playground-evaluation/index.ts | 2 + .../manual-test-cases.ts | 86 +++++++ testing/mocks/graphql-client.ts | 5 + testing/mocks/index.ts | 2 + testing/mocks/opub-ui.tsx | 224 ++++++++++++++++++ testing/setup.ts | 33 ++- 25 files changed, 2137 insertions(+), 1 deletion(-) create mode 100644 features/ai-maker/components/manual-evaluation/CompletedTestCases.test.tsx create mode 100644 features/ai-maker/components/manual-evaluation/EvaluateOutputSection.test.ts create mode 100644 features/ai-maker/components/manual-evaluation/ManualTestCaseDetailSheet.test.tsx create mode 100644 features/ai-maker/components/manual-evaluation/RecommendationModal.test.tsx create mode 100644 features/ai-maker/components/manual-evaluation/utils.test.ts create mode 100644 features/ai-maker/utils/map-audit-results.test.ts create mode 100644 features/dashboard/components/AddIssueModal.test.tsx create mode 100644 features/dashboard/components/AuditResultsList.test.tsx create mode 100644 features/dashboard/components/BulkTestCaseDetailSheet.test.tsx create mode 100644 features/dashboard/components/EvaluationDetail/hooks/use-audit-actions.test.ts create mode 100644 features/dashboard/components/EvaluationDetail/hooks/use-audit-data.test.ts create mode 100644 features/dashboard/components/EvaluationDetail/hooks/use-audit-polling.test.ts create mode 100644 features/dashboard/components/EvaluationDetail/hooks/use-evaluation-detail.test.ts create mode 100644 features/dashboard/utils/evaluation.test.ts create mode 100644 testing/fixtures/bulk-evaluation/audit-results.ts create mode 100644 testing/fixtures/bulk-evaluation/audit.ts create mode 100644 testing/fixtures/bulk-evaluation/index.ts create mode 100644 testing/fixtures/playground-evaluation/audit.ts create mode 100644 testing/fixtures/playground-evaluation/index.ts create mode 100644 testing/fixtures/playground-evaluation/manual-test-cases.ts create mode 100644 testing/mocks/graphql-client.ts create mode 100644 testing/mocks/index.ts create mode 100644 testing/mocks/opub-ui.tsx diff --git a/features/ai-maker/components/manual-evaluation/CompletedTestCases.test.tsx b/features/ai-maker/components/manual-evaluation/CompletedTestCases.test.tsx new file mode 100644 index 0000000..9c610ff --- /dev/null +++ b/features/ai-maker/components/manual-evaluation/CompletedTestCases.test.tsx @@ -0,0 +1,122 @@ +import userEvent from '@testing-library/user-event'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { + makeManualTestCase, + makePassingManualTestCase, +} from '@/testing/fixtures/playground-evaluation'; +import { render, screen } from '@/testing/utils'; +import CompletedTestCases from './CompletedTestCases'; + +vi.mock('opub-ui', async () => { + const { getOpubUiMockModule } = await import('@/testing/mocks/opub-ui'); + return getOpubUiMockModule(); +}); + +vi.mock('./ManualTestCaseDetailSheet', () => ({ + default: ({ open, testCase }: { open: boolean; testCase: { displayIndex: number } | null }) => + open && testCase ? ( +
Detail for Input {testCase.displayIndex}
+ ) : null, +})); + +vi.mock('react-markdown', () => ({ + default: ({ children }: { children: string }) =>
{children}
, +})); + +vi.mock('@tabler/icons-react', () => ({ + IconArrowsDiagonal: () => , +})); + +describe('CompletedTestCases', () => { + const subModules = [{ name: 'GENDER_BIAS', displayName: 'Gender Bias' }]; + const getModuleDisplayName = (name: string) => name; + + beforeEach(() => { + vi.clearAllMocks(); + }); + + it('returns null when there are no test cases', () => { + const { container } = render( + + ); + + expect(container).toBeEmptyDOMElement(); + }); + + it('renders failed test case cards with risk tags', () => { + render( + + ); + + expect(screen.getByText('Completed Test Cases')).toBeInTheDocument(); + expect(screen.getByText('What is the role of women in society?')).toBeInTheDocument(); + expect(screen.getByText('High risk - Gender Bias')).toBeInTheDocument(); + }); + + it('renders passed tag for passing test cases', () => { + render( + + ); + + expect(screen.getByText('Passed')).toBeInTheDocument(); + }); + + it('sorts test cases by createdAt ascending', () => { + render( + + ); + + const cards = screen.getAllByRole('button', { name: /View details for input/ }); + expect(cards[0]).toHaveAccessibleName('View details for input 1'); + expect(screen.getByText('Earlier input')).toBeInTheDocument(); + expect(screen.getByText('Later input')).toBeInTheDocument(); + }); + + it('opens detail sheet when a card is clicked', async () => { + const user = userEvent.setup(); + + render( + + ); + + await user.click(screen.getByRole('button', { name: 'View details for input 1' })); + + expect(screen.getByTestId('manual-detail-sheet')).toHaveTextContent('Detail for Input 1'); + }); +}); diff --git a/features/ai-maker/components/manual-evaluation/EvaluateOutputSection.test.ts b/features/ai-maker/components/manual-evaluation/EvaluateOutputSection.test.ts new file mode 100644 index 0000000..6293d94 --- /dev/null +++ b/features/ai-maker/components/manual-evaluation/EvaluateOutputSection.test.ts @@ -0,0 +1,14 @@ +import { describe, expect, it, vi } from 'vitest'; +import { createEvaluationIssueRow } from './EvaluateOutputSection'; + +describe('createEvaluationIssueRow', () => { + it('creates an empty issue row with a generated id', () => { + const row = createEvaluationIssueRow(); + + expect(row.id).toMatch(/^issue-\d+-[a-z0-9]+$/); + expect(row.issueType).toBe(''); + expect(row.severity).toBe(''); + expect(row.observations).toBe(''); + expect(row.idealOutput).toBe(''); + }); +}); diff --git a/features/ai-maker/components/manual-evaluation/ManualTestCaseDetailSheet.test.tsx b/features/ai-maker/components/manual-evaluation/ManualTestCaseDetailSheet.test.tsx new file mode 100644 index 0000000..127189d --- /dev/null +++ b/features/ai-maker/components/manual-evaluation/ManualTestCaseDetailSheet.test.tsx @@ -0,0 +1,72 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { + makeManualTestCase, + makePassingManualTestCase, +} from '@/testing/fixtures/playground-evaluation'; +import { render, screen } from '@/testing/utils'; +import ManualTestCaseDetailSheet from './ManualTestCaseDetailSheet'; + +vi.mock('opub-ui', async () => { + const { getOpubUiMockModule } = await import('@/testing/mocks/opub-ui'); + return getOpubUiMockModule(); +}); + +vi.mock('@/components/icons', () => ({ + Icons: { cross: 'cross', info: 'info' }, +})); + +vi.mock('react-markdown', () => ({ + default: ({ children }: { children: string }) =>
{children}
, +})); + +describe('ManualTestCaseDetailSheet', () => { + const subModules = [{ name: 'GENDER_BIAS', displayName: 'Gender Bias' }]; + + beforeEach(() => { + vi.clearAllMocks(); + }); + + it('does not render when closed', () => { + render( + + ); + + expect(screen.queryByRole('dialog')).not.toBeInTheDocument(); + }); + + it('renders failed issues for a test case', () => { + render( + + ); + + expect(screen.getByText('Input 1')).toBeInTheDocument(); + expect(screen.getByText('What is the role of women in society?')).toBeInTheDocument(); + expect(screen.getByText('Women should stay at home.')).toBeInTheDocument(); + expect(screen.getByText('High risk - Gender Bias')).toBeInTheDocument(); + expect(screen.getByText('Biased output detected')).toBeInTheDocument(); + }); + + it('renders passed state when test case has no failed issues', () => { + render( + + ); + + expect(screen.getByText('Input 2')).toBeInTheDocument(); + expect(screen.getByText('Passed')).toBeInTheDocument(); + }); +}); diff --git a/features/ai-maker/components/manual-evaluation/RecommendationModal.test.tsx b/features/ai-maker/components/manual-evaluation/RecommendationModal.test.tsx new file mode 100644 index 0000000..e261be5 --- /dev/null +++ b/features/ai-maker/components/manual-evaluation/RecommendationModal.test.tsx @@ -0,0 +1,73 @@ +import userEvent from '@testing-library/user-event'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { render, screen } from '@/testing/utils'; +import RecommendationModal from './RecommendationModal'; + +vi.mock('opub-ui', async () => { + const { getOpubUiMockModule } = await import('@/testing/mocks/opub-ui'); + return getOpubUiMockModule(); +}); + +describe('RecommendationModal', () => { + const onSubmit = vi.fn(); + const onOpenChange = vi.fn(); + + beforeEach(() => { + vi.clearAllMocks(); + }); + + const renderModal = (open = true) => + render( + + ); + + it('does not render when closed', () => { + renderModal(false); + expect(screen.queryByRole('dialog')).not.toBeInTheDocument(); + }); + + it('disables submit until recommendation text is entered', () => { + renderModal(); + expect(screen.getByRole('button', { name: 'Submit' })).toBeDisabled(); + }); + + it('keeps submit disabled for whitespace-only recommendation', async () => { + const user = userEvent.setup(); + renderModal(); + + await user.type(screen.getByPlaceholderText('Enter your recommendation...'), ' '); + + expect(screen.getByRole('button', { name: 'Submit' })).toBeDisabled(); + expect(onSubmit).not.toHaveBeenCalled(); + }); + + it('submits trimmed recommendation and closes modal', async () => { + const user = userEvent.setup(); + renderModal(); + + await user.type( + screen.getByPlaceholderText('Enter your recommendation...'), + ' Model needs bias mitigation ' + ); + await user.click(screen.getByRole('button', { name: 'Submit' })); + + expect(onSubmit).toHaveBeenCalledWith('Model needs bias mitigation'); + expect(onOpenChange).toHaveBeenCalledWith(false); + }); + + it('clears form on cancel', async () => { + const user = userEvent.setup(); + renderModal(); + + const textarea = screen.getByPlaceholderText('Enter your recommendation...'); + await user.type(textarea, 'Draft recommendation'); + await user.click(screen.getByRole('button', { name: 'Cancel' })); + + expect(onOpenChange).toHaveBeenCalledWith(false); + }); +}); diff --git a/features/ai-maker/components/manual-evaluation/utils.test.ts b/features/ai-maker/components/manual-evaluation/utils.test.ts new file mode 100644 index 0000000..99f198d --- /dev/null +++ b/features/ai-maker/components/manual-evaluation/utils.test.ts @@ -0,0 +1,199 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { + makeManualTestCase, + makeModuleProgress, + makeWorkspaceDraft, +} from '@/testing/fixtures/playground-evaluation'; +import type { ManualTestCaseIssue } from './types'; +import { + clearManualEvalWorkspaceDraft, + formatRiskLabel, + getFailedManualTestCaseIssues, + getFallbackEvaluationModules, + getFallbackSubModules, + getIssueRiskTagColors, + getManualEvalWorkspaceStorageKey, + getTotalManualTestCaseCount, + isManualTestCasePassed, + metricsToSubModules, + MIN_PLAYGROUND_TEST_CASES, + normalizeIssueSeverity, + readManualEvalWorkspaceDraft, + resolveIssueDisplayName, + writeManualEvalWorkspaceDraft, +} from './utils'; + +describe('normalizeIssueSeverity', () => { + it('maps HIGH, MEDIUM, and LOW', () => { + expect(normalizeIssueSeverity('HIGH')).toBe('HIGH'); + expect(normalizeIssueSeverity('medium_risk')).toBe('MEDIUM'); + expect(normalizeIssueSeverity('low')).toBe('LOW'); + }); + + it('returns null for empty or unknown values', () => { + expect(normalizeIssueSeverity('')).toBeNull(); + expect(normalizeIssueSeverity(null)).toBeNull(); + expect(normalizeIssueSeverity('UNKNOWN')).toBeNull(); + }); +}); + +describe('getIssueRiskTagColors', () => { + it('returns severity-specific colors', () => { + expect(getIssueRiskTagColors('HIGH').textColor).toBe('#E11D48'); + expect(getIssueRiskTagColors('MEDIUM').textColor).toBe('#92400E'); + expect(getIssueRiskTagColors('LOW').textColor).toBe('#2563EB'); + }); + + it('returns default colors for unknown severity', () => { + expect(getIssueRiskTagColors('UNKNOWN').fillColor).toBe('#F3F4F6'); + }); +}); + +describe('formatRiskLabel', () => { + it('formats severity and label together', () => { + expect(formatRiskLabel('HIGH', 'Gender Bias')).toBe('High risk - Gender Bias'); + }); + + it('returns label only when severity is missing', () => { + expect(formatRiskLabel(null, 'Gender Bias')).toBe('Gender Bias'); + }); +}); + +describe('getFailedManualTestCaseIssues', () => { + it('excludes PASSED issues', () => { + const issues: ManualTestCaseIssue[] = [ + { metricName: 'GENDER_BIAS', status: 'PASSED' }, + { metricName: 'CASTE_BIAS', status: 'FAILED', severity: 'HIGH' }, + ]; + expect(getFailedManualTestCaseIssues(issues)).toHaveLength(1); + }); + + it('treats issues with content as failed when status is not PASSED', () => { + const issues: ManualTestCaseIssue[] = [ + { metricName: '', status: '', severity: 'LOW', comments: 'Issue noted' }, + ]; + expect(getFailedManualTestCaseIssues(issues)).toHaveLength(1); + }); + + it('ignores empty non-failed issues', () => { + const issues: ManualTestCaseIssue[] = [{ metricName: '', status: '' }]; + expect(getFailedManualTestCaseIssues(issues)).toHaveLength(0); + }); +}); + +describe('isManualTestCasePassed', () => { + it('returns true when no failed issues remain', () => { + expect(isManualTestCasePassed(makeManualTestCase({ issues: [] }))).toBe(true); + expect( + isManualTestCasePassed( + makeManualTestCase({ issues: [{ metricName: 'GENDER_BIAS', status: 'PASSED' }] }) + ) + ).toBe(true); + }); + + it('returns false when failed issues exist', () => { + expect(isManualTestCasePassed(makeManualTestCase())).toBe(false); + }); +}); + +describe('workspace draft sessionStorage helpers', () => { + const orgId = 'org-1'; + const auditId = 'audit-1'; + + beforeEach(() => { + sessionStorage.clear(); + }); + + it('builds a stable storage key', () => { + expect(getManualEvalWorkspaceStorageKey(orgId, auditId)).toBe( + 'manual-eval-workspace:org-1:audit-1' + ); + }); + + it('round-trips a valid draft', () => { + const draft = makeWorkspaceDraft(); + writeManualEvalWorkspaceDraft(orgId, auditId, draft); + + expect(readManualEvalWorkspaceDraft(orgId, auditId)).toEqual(draft); + }); + + it('returns null for malformed JSON', () => { + sessionStorage.setItem(getManualEvalWorkspaceStorageKey(orgId, auditId), '{bad json'); + expect(readManualEvalWorkspaceDraft(orgId, auditId)).toBeNull(); + }); + + it('sanitizes invalid draft fields', () => { + sessionStorage.setItem( + getManualEvalWorkspaceStorageKey(orgId, auditId), + JSON.stringify({ + selectedModule: 123, + sourceLanguage: 'en', + targetLanguage: 'hi', + inputPrompt: 'Prompt', + modelOutput: 'Output', + hasCalledModel: true, + status: 'INVALID', + issueRows: [{ id: 'row-1', issueType: 'GENDER_BIAS' }], + }) + ); + + const draft = readManualEvalWorkspaceDraft(orgId, auditId); + expect(draft?.selectedModule).toBeNull(); + expect(draft?.status).toBeNull(); + expect(draft?.issueRows).toEqual([]); + }); + + it('clears stored draft', () => { + writeManualEvalWorkspaceDraft(orgId, auditId, makeWorkspaceDraft()); + clearManualEvalWorkspaceDraft(orgId, auditId); + expect(readManualEvalWorkspaceDraft(orgId, auditId)).toBeNull(); + }); +}); + +describe('module helpers', () => { + it('exposes fallback modules and sub-modules', () => { + const modules = getFallbackEvaluationModules(); + expect(modules.map((module) => module.name)).toContain('BIAS_FAIRNESS'); + expect(getFallbackSubModules('BIAS_FAIRNESS').length).toBeGreaterThan(0); + }); + + it('maps metric options to sub-modules', () => { + expect( + metricsToSubModules([ + { value: 'GENDER_BIAS', label: 'Gender Bias' }, + { value: '', label: 'Ignored' }, + ]) + ).toEqual([{ name: 'GENDER_BIAS', displayName: 'Gender Bias' }]); + }); + + it('sums manual test case counts from module progress', () => { + expect( + getTotalManualTestCaseCount([ + makeModuleProgress({ testCaseCount: 2 }), + makeModuleProgress({ testCaseCount: 3 }), + ]) + ).toBe(5); + }); +}); + +describe('resolveIssueDisplayName', () => { + const subModules = [{ name: 'GENDER_BIAS', displayName: 'Gender Bias' }]; + + it('prefers API display names', () => { + expect(resolveIssueDisplayName('GENDER_BIAS', subModules)).toBe('Gender Bias'); + }); + + it('falls back to module-specific static labels', () => { + expect(resolveIssueDisplayName('CASTE_BIAS', [], 'BIAS_FAIRNESS')).toBe('Caste Bias'); + }); + + it('humanizes unknown keys', () => { + expect(resolveIssueDisplayName('CUSTOM_ISSUE', [])).toBe('Custom Issue'); + }); +}); + +describe('MIN_PLAYGROUND_TEST_CASES', () => { + it('requires at least three test cases to finish', () => { + expect(MIN_PLAYGROUND_TEST_CASES).toBe(3); + }); +}); diff --git a/features/ai-maker/utils/map-audit-results.test.ts b/features/ai-maker/utils/map-audit-results.test.ts new file mode 100644 index 0000000..a357a79 --- /dev/null +++ b/features/ai-maker/utils/map-audit-results.test.ts @@ -0,0 +1,149 @@ +import { beforeEach, describe, expect, it } from 'vitest'; +import { + makeAuditResult, + makeGroupedAuditResults, + makePassingResult, + makeReviewedPassingResult, + makeReviewedResult, + resetFixtureCounter, +} from '@/testing/fixtures/bulk-evaluation'; +import { isIssueResult, mapAuditResultsToBulkTestCases, mapRiskLevel } from './map-audit-results'; + +describe('mapRiskLevel', () => { + it('maps HIGH, MEDIUM, and LOW risk levels', () => { + expect(mapRiskLevel('HIGH')).toBe('HIGH'); + expect(mapRiskLevel('MEDIUM')).toBe('MEDIUM'); + expect(mapRiskLevel('LOW')).toBe('LOW'); + }); + + it('maps partial risk level strings', () => { + expect(mapRiskLevel('HIGH_RISK')).toBe('HIGH'); + expect(mapRiskLevel('medium_risk')).toBe('MEDIUM'); + expect(mapRiskLevel('low risk')).toBe('LOW'); + }); + + it('returns null for no-risk levels', () => { + expect(mapRiskLevel('NO_RISK')).toBeNull(); + expect(mapRiskLevel('NONE')).toBeNull(); + expect(mapRiskLevel('NO RISK')).toBeNull(); + expect(mapRiskLevel('')).toBeNull(); + }); + + it('returns null for null and undefined', () => { + expect(mapRiskLevel(null)).toBeNull(); + expect(mapRiskLevel(undefined)).toBeNull(); + }); + + it('is case-insensitive', () => { + expect(mapRiskLevel('high')).toBe('HIGH'); + expect(mapRiskLevel('Medium')).toBe('MEDIUM'); + }); +}); + +describe('isIssueResult', () => { + beforeEach(() => { + resetFixtureCounter(); + }); + + it('returns false when success is true', () => { + expect(isIssueResult(makePassingResult())).toBe(false); + }); + + it('returns true for failed results with reason', () => { + expect(isIssueResult(makeAuditResult({ success: false, reason: 'Bias detected' }))).toBe(true); + }); + + it('returns false for reviewed results marked as passing', () => { + expect(isIssueResult(makeReviewedPassingResult())).toBe(false); + }); + + it('returns true for reviewed results with evaluator risk', () => { + expect(isIssueResult(makeReviewedResult())).toBe(true); + }); + + it('returns false for reviewed NO_RISK evaluator level', () => { + expect( + isIssueResult( + makeReviewedResult({ + evaluatorSuccess: false, + evaluatorRiskLevel: 'NO_RISK', + evaluatorReason: '', + }) + ) + ).toBe(false); + }); +}); + +describe('mapAuditResultsToBulkTestCases', () => { + beforeEach(() => { + resetFixtureCounter(); + }); + + it('returns empty arrays for empty input', () => { + expect(mapAuditResultsToBulkTestCases([])).toEqual({ + items: [], + moduleIssueCounts: [], + }); + }); + + it('skips results without a test id', () => { + const result = makeAuditResult(); + result.task!.test!.id = undefined as unknown as string; + + const { items } = mapAuditResultsToBulkTestCases([result]); + expect(items).toHaveLength(0); + }); + + it('groups results by test id', () => { + const grouped = makeGroupedAuditResults('test-group-1'); + const { items } = mapAuditResultsToBulkTestCases(grouped); + + expect(items).toHaveLength(1); + expect(items[0].id).toBe('test-group-1'); + expect(items[0].risks).toHaveLength(2); + expect(items[0].allMetricResults).toHaveLength(2); + }); + + it('uses first non-empty line as input prompt', () => { + const grouped = makeGroupedAuditResults('test-prompt'); + const { items } = mapAuditResultsToBulkTestCases(grouped); + + expect(items[0].inputPrompt).toBe('Line one'); + expect(items[0].fullInputText).toContain('Line two prompt'); + }); + + it('builds module issue counts from issue results', () => { + const grouped = makeGroupedAuditResults('test-modules'); + const { moduleIssueCounts } = mapAuditResultsToBulkTestCases(grouped); + + expect(moduleIssueCounts).toHaveLength(2); + expect(moduleIssueCounts.find((m) => m.moduleId === 'BIAS_FAIRNESS')?.issueCount).toBe(1); + expect(moduleIssueCounts.find((m) => m.moduleId === 'PRIVACY_SAFETY')?.issueCount).toBe(1); + }); + + it('handles reviewed vs unreviewed risk paths', () => { + const unreviewed = makeAuditResult({ testId: 'test-a', isReviewed: false }); + const reviewed = makeReviewedResult({ testId: 'test-b' }); + + const { items } = mapAuditResultsToBulkTestCases([unreviewed, reviewed]); + + expect(items).toHaveLength(2); + expect(items[0].risks[0].observation).toBe('Test failure reason'); + expect(items[1].risks[0].observation).toBe('Reviewer noted bias'); + }); + + it('defaults missing text fields to em dash', () => { + const result = makeAuditResult({ + testId: 'empty-text', + task: { + id: 'task-empty', + module: 'BIAS_FAIRNESS', + test: { id: 'empty-text', testInput: '', actualOutput: '' }, + }, + }); + + const { items } = mapAuditResultsToBulkTestCases([result]); + expect(items[0].inputPrompt).toBe('—'); + expect(items[0].output).toBe('—'); + }); +}); diff --git a/features/dashboard/components/AddIssueModal.test.tsx b/features/dashboard/components/AddIssueModal.test.tsx new file mode 100644 index 0000000..42461c7 --- /dev/null +++ b/features/dashboard/components/AddIssueModal.test.tsx @@ -0,0 +1,114 @@ +import userEvent from '@testing-library/user-event'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { render, screen } from '@/testing/utils'; +import AddIssueModal from './AddIssueModal'; + +vi.mock('opub-ui', async () => { + const { getOpubUiMockModule } = await import('@/testing/mocks/opub-ui'); + return getOpubUiMockModule(); +}); + +const issueOptions = [ + { value: 'result-1', label: 'Gender Bias' }, + { value: 'result-2', label: 'Safety Risk' }, +]; + +const severityOptions = [ + { value: 'HIGH', label: 'High' }, + { value: 'MEDIUM', label: 'Medium' }, + { value: 'LOW', label: 'Low' }, +]; + +describe('AddIssueModal', () => { + const onSubmit = vi.fn(); + const onOpenChange = vi.fn(); + + beforeEach(() => { + vi.clearAllMocks(); + }); + + const renderModal = (open = true) => + render( + + ); + + it('does not render when closed', () => { + renderModal(false); + expect(screen.queryByRole('dialog')).not.toBeInTheDocument(); + }); + + it('shows validation errors when submitting empty form', async () => { + const user = userEvent.setup(); + renderModal(); + + await user.click(screen.getByRole('button', { name: 'Save' })); + + expect(screen.getByText('Issue is required')).toBeInTheDocument(); + expect(screen.getByText('Risk level is required')).toBeInTheDocument(); + expect(screen.getByText('Reasons or observations are required')).toBeInTheDocument(); + expect(onSubmit).not.toHaveBeenCalled(); + }); + + it('submits form data when all fields are valid', async () => { + const user = userEvent.setup(); + renderModal(); + + await user.selectOptions(screen.getByLabelText('Issue'), 'result-1'); + await user.selectOptions(screen.getByLabelText('Risk Level'), 'HIGH'); + await user.type( + screen.getByLabelText('Reasons or Observations'), + 'Observed gender bias in output' + ); + await user.click(screen.getByRole('button', { name: 'Save' })); + + expect(onSubmit).toHaveBeenCalledWith({ + resultId: 'result-1', + label: 'Gender Bias', + severity: 'HIGH', + observation: 'Observed gender bias in output', + }); + expect(onOpenChange).toHaveBeenCalledWith(false); + }); + + it('closes modal on cancel', async () => { + const user = userEvent.setup(); + renderModal(); + + await user.click(screen.getByRole('button', { name: 'Cancel' })); + + expect(onOpenChange).toHaveBeenCalledWith(false); + }); + + it('resets form when modal closes', async () => { + const user = userEvent.setup(); + const { rerender } = renderModal(); + + await user.type(screen.getByLabelText('Reasons or Observations'), 'Draft text'); + rerender( + + ); + rerender( + + ); + + expect(screen.getByLabelText('Reasons or Observations')).toHaveValue(''); + }); +}); diff --git a/features/dashboard/components/AuditResultsList.test.tsx b/features/dashboard/components/AuditResultsList.test.tsx new file mode 100644 index 0000000..665194f --- /dev/null +++ b/features/dashboard/components/AuditResultsList.test.tsx @@ -0,0 +1,142 @@ +import userEvent from '@testing-library/user-event'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { + makeAuditResult, + makeGroupedAuditResults, + resetFixtureCounter, +} from '@/testing/fixtures/bulk-evaluation'; +import { render, screen } from '@/testing/utils'; +import AuditResultsList from './AuditResultsList'; + +vi.mock('opub-ui', async () => { + const { getOpubUiMockModule } = await import('@/testing/mocks/opub-ui'); + return getOpubUiMockModule(); +}); + +vi.mock('./BulkTestCaseDetailSheet', () => ({ + default: ({ open, testCase }: { open: boolean; testCase: { index: number } | null }) => + open && testCase ? ( +
Detail for Input {testCase.index}
+ ) : null, +})); + +vi.mock('react-markdown', () => ({ + default: ({ children }: { children: string }) =>
{children}
, +})); + +vi.mock('@tabler/icons-react', () => ({ + IconArrowsDiagonal: () => , + IconSparkles: () => , +})); + +describe('AuditResultsList', () => { + const metricSummary = {}; + + beforeEach(() => { + resetFixtureCounter(); + vi.clearAllMocks(); + }); + + it('shows loading spinner when results are null', () => { + render(); + + expect(screen.getByRole('status')).toHaveTextContent('Loading'); + }); + + it('renders test case cards from audit results', () => { + const results = makeGroupedAuditResults('test-1'); + + render(); + + expect(screen.getByText('Input 1')).toBeInTheDocument(); + expect(screen.getByText('Line one')).toBeInTheDocument(); + expect(screen.getByText('Output text')).toBeInTheDocument(); + expect(screen.getByText(/High risk - Gender Bias/)).toBeInTheDocument(); + }); + + it('sorts by issue count high to low by default', () => { + resetFixtureCounter(); + const lowIssues = makeAuditResult({ + testId: 'test-low', + riskLevel: 'LOW', + reason: 'Minor issue', + task: { + id: 'task-low', + module: 'BIAS_FAIRNESS', + metric: 'metric_a', + test: { id: 'test-low', testInput: 'Low issues input', actualOutput: 'Out' }, + }, + }); + const highIssues = makeGroupedAuditResults('test-high'); + + render( + + ); + + const cards = screen.getAllByRole('button', { name: /View details for input/ }); + expect(cards[0]).toHaveAccessibleName('View details for input 2'); + expect(cards[1]).toHaveAccessibleName('View details for input 1'); + }); + + it('sorts by issue count low to high when selected', async () => { + const user = userEvent.setup(); + resetFixtureCounter(); + const lowIssues = makeAuditResult({ + testId: 'test-low', + riskLevel: 'LOW', + reason: 'Minor issue', + task: { + id: 'task-low', + module: 'BIAS_FAIRNESS', + metric: 'metric_a', + test: { id: 'test-low', testInput: 'Low issues input', actualOutput: 'Out' }, + }, + }); + const highIssues = makeGroupedAuditResults('test-high'); + + render( + + ); + + await user.selectOptions(screen.getByLabelText('Sort'), 'issues_asc'); + + const cards = screen.getAllByRole('button', { name: /View details for input/ }); + expect(cards[0]).toHaveAccessibleName('View details for input 1'); + expect(cards[1]).toHaveAccessibleName('View details for input 2'); + }); + + it('opens detail sheet when a card is clicked', async () => { + const user = userEvent.setup(); + const results = makeGroupedAuditResults('test-open'); + + render(); + + const card = screen.getByRole('button', { name: 'View details for input 1' }); + await user.click(card); + + expect(screen.getByTestId('detail-sheet')).toHaveTextContent('Detail for Input 1'); + }); + + it('shows empty state when no inputs match', () => { + render(); + + expect(screen.getByText('No inputs found for the selected module.')).toBeInTheDocument(); + }); + + it('shows module issue pills', () => { + const results = makeGroupedAuditResults('test-pills'); + + render(); + + const pills = screen.getByText(/Bias and Fairness - 2 Issues/); + expect(pills).toBeInTheDocument(); + }); +}); diff --git a/features/dashboard/components/BulkTestCaseDetailSheet.test.tsx b/features/dashboard/components/BulkTestCaseDetailSheet.test.tsx new file mode 100644 index 0000000..8b8f73a --- /dev/null +++ b/features/dashboard/components/BulkTestCaseDetailSheet.test.tsx @@ -0,0 +1,109 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { UPDATE_AUDIT_RESULT_MUTATION } from '@/features/ai-maker/api/bulk-evaluation-queries'; +import { makeBulkTestCase } from '@/testing/fixtures/bulk-evaluation'; +import { fireEvent, render, screen, waitFor } from '@/testing/utils'; +import BulkTestCaseDetailSheet from './BulkTestCaseDetailSheet'; + +const mockRequest = vi.fn(); + +vi.mock('@/lib/graphql-client', () => ({ + useGraphQL: () => ({ request: mockRequest }), +})); + +vi.mock('opub-ui', async () => { + const { getOpubUiMockModule } = await import('@/testing/mocks/opub-ui'); + return getOpubUiMockModule(); +}); + +vi.mock('@/components/icons', () => ({ + Icons: { cross: 'cross', info: 'info' }, +})); + +vi.mock('react-markdown', () => ({ + default: ({ children }: { children: string }) =>
{children}
, +})); + +vi.mock('@tabler/icons-react', () => ({ + IconTrash: () => , +})); + +describe('BulkTestCaseDetailSheet', () => { + const testCase = makeBulkTestCase(); + + beforeEach(() => { + vi.clearAllMocks(); + mockRequest.mockResolvedValue({ + updateAuditResult: { success: true }, + }); + }); + + it('does not render content when closed', () => { + render(); + + expect(screen.queryByRole('dialog')).not.toBeInTheDocument(); + }); + + it('renders test case risks and content when open', () => { + render(); + + expect(screen.getByText('Input 1')).toBeInTheDocument(); + expect(screen.getByText(testCase.fullInputText)).toBeInTheDocument(); + expect(screen.getByText(testCase.output)).toBeInTheDocument(); + expect(screen.getByDisplayValue('Biased response detected')).toBeInTheDocument(); + }); + + it('shows no issues message when risks are empty', () => { + render( + + ); + + expect(screen.getByText('No issues identified for this input.')).toBeInTheDocument(); + }); + + it('triggers save when severity changes in editable mode', async () => { + render( + + ); + + fireEvent.change(screen.getByLabelText('Severity'), { target: { value: 'MEDIUM' } }); + + await waitFor(() => { + expect(mockRequest).toHaveBeenCalledWith( + UPDATE_AUDIT_RESULT_MUTATION, + { + input: { + resultId: 'result-1', + evaluatorRiskLevel: 'MEDIUM_RISK', + }, + }, + { organization: 'org-1' } + ); + }); + }); + + it('opens add issue modal when editable and metrics allow', async () => { + render( + + ); + + fireEvent.click(screen.getByRole('button', { name: 'Add an issue' })); + + expect(screen.getByRole('dialog', { name: 'Add an Issue' })).toBeInTheDocument(); + }); +}); diff --git a/features/dashboard/components/EvaluationDetail/hooks/use-audit-actions.test.ts b/features/dashboard/components/EvaluationDetail/hooks/use-audit-actions.test.ts new file mode 100644 index 0000000..ba76610 --- /dev/null +++ b/features/dashboard/components/EvaluationDetail/hooks/use-audit-actions.test.ts @@ -0,0 +1,150 @@ +import { act, renderHook, waitFor } from '@testing-library/react'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { SUBMIT_AUDIT_REVIEW_MUTATION } from '@/features/dashboard/api/evaluation-queries'; +import { makeBulkAudit } from '@/testing/fixtures/bulk-evaluation'; +import { mockToast } from '@/testing/mocks/opub-ui'; +import { useAuditActions } from './use-audit-actions'; + +const mockRequest = vi.fn(); +const mockStopProgressPolling = vi.fn(); +const mockFetchAuditSummary = vi.fn(); +const mockSetAudit = vi.fn(); +const mockSetIsEvaluationSaved = vi.fn(); + +vi.mock('@/lib/graphql-client', () => ({ + useGraphQL: () => ({ request: mockRequest }), +})); + +vi.mock('opub-ui', async () => { + const { getOpubUiMockModule } = await import('@/testing/mocks/opub-ui'); + return getOpubUiMockModule(); +}); + +describe('useAuditActions', () => { + beforeEach(() => { + vi.clearAllMocks(); + mockFetchAuditSummary.mockResolvedValue({ hasReport: false }); + mockSetAudit.mockImplementation((updater) => { + if (typeof updater === 'function') { + updater(makeBulkAudit()); + } + }); + }); + + const renderActions = (audit = makeBulkAudit()) => + renderHook(() => + useAuditActions({ + evaluationId: 'eval-1', + orgId: 'org-1', + audit, + auditReport: null, + setAudit: mockSetAudit, + setIsEvaluationSaved: mockSetIsEvaluationSaved, + fetchAuditSummary: mockFetchAuditSummary, + stopProgressPolling: mockStopProgressPolling, + }) + ); + + describe('submitBulkReview', () => { + it('does not call GraphQL when status is not PENDING_REVIEW', async () => { + const audit = makeBulkAudit({ status: 'IN_PROGRESS' }); + const { result } = renderActions(audit); + + await act(async () => { + await result.current.submitBulkReview('Recommendation text'); + }); + + expect(mockRequest).not.toHaveBeenCalled(); + }); + + it('submits review successfully and updates audit state', async () => { + mockRequest.mockResolvedValue({ + submitAuditReview: { + success: true, + audit: { + id: 'audit-1', + status: 'COMPLETED', + completedAt: '2026-01-02T00:00:00Z', + }, + }, + }); + + const { result } = renderActions(); + + await act(async () => { + await result.current.submitBulkReview('Final recommendation'); + }); + + expect(mockRequest).toHaveBeenCalledWith( + SUBMIT_AUDIT_REVIEW_MUTATION, + { + input: { + auditId: 'audit-1', + recommendations: 'Final recommendation', + }, + }, + { organization: 'org-1' } + ); + expect(mockToast.success).toHaveBeenCalledWith('Review submitted successfully.'); + expect(mockSetIsEvaluationSaved).toHaveBeenCalledWith(true); + expect(mockStopProgressPolling).toHaveBeenCalled(); + expect(mockFetchAuditSummary).toHaveBeenCalled(); + }); + + it('sends null recommendations when recommendation is empty', async () => { + mockRequest.mockResolvedValue({ + submitAuditReview: { + success: true, + audit: { id: 'audit-1', status: 'COMPLETED', completedAt: null }, + }, + }); + + const { result } = renderActions(); + + await act(async () => { + await result.current.submitBulkReview(' '); + }); + + expect(mockRequest).toHaveBeenCalledWith( + SUBMIT_AUDIT_REVIEW_MUTATION, + { + input: { + auditId: 'audit-1', + recommendations: null, + }, + }, + { organization: 'org-1' } + ); + }); + + it('shows error toast on failure without updating audit', async () => { + mockRequest.mockResolvedValue({ + submitAuditReview: { success: false, message: 'Server error' }, + }); + + const { result } = renderActions(); + + await act(async () => { + await result.current.submitBulkReview('Recommendation'); + }); + + expect(mockToast.error).toHaveBeenCalledWith('Server error'); + expect(mockSetIsEvaluationSaved).not.toHaveBeenCalled(); + expect(mockStopProgressPolling).not.toHaveBeenCalled(); + }); + + it('shows error toast on request exception', async () => { + mockRequest.mockRejectedValue(new Error('Network error')); + + const { result } = renderActions(); + + await act(async () => { + await result.current.submitBulkReview('Recommendation'); + }); + + await waitFor(() => { + expect(mockToast.error).toHaveBeenCalledWith('Network error'); + }); + }); + }); +}); diff --git a/features/dashboard/components/EvaluationDetail/hooks/use-audit-data.test.ts b/features/dashboard/components/EvaluationDetail/hooks/use-audit-data.test.ts new file mode 100644 index 0000000..1b42f1f --- /dev/null +++ b/features/dashboard/components/EvaluationDetail/hooks/use-audit-data.test.ts @@ -0,0 +1,108 @@ +import { renderHook, waitFor } from '@testing-library/react'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { + GET_AUDIT_QUERY, + GET_AUDIT_RESULTS_QUERY, + GET_AUDIT_SUMMARY, +} from '@/features/dashboard/api/evaluation-queries'; +import { makePlaygroundAudit } from '@/testing/fixtures/playground-evaluation'; +import { useAuditData } from './use-audit-data'; + +const mockRequest = vi.fn(); + +vi.mock('@/lib/graphql-client', () => ({ + useGraphQL: () => ({ + request: mockRequest, + isAuthenticated: true, + isLoading: false, + }), +})); + +describe('useAuditData playground behavior', () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + it('does not fetch results while playground audit is in progress', async () => { + mockRequest.mockImplementation((query: string) => { + if (query === GET_AUDIT_QUERY) { + return Promise.resolve({ + audit: makePlaygroundAudit({ status: 'IN_PROGRESS', evaluationMode: 'PLAYGROUND' }), + }); + } + return Promise.resolve({}); + }); + + const { result } = renderHook(() => useAuditData('eval-1', 'org-1')); + + await waitFor(() => { + expect(result.current.isLoading).toBe(false); + }); + + expect(result.current.audit?.evaluationMode).toBe('PLAYGROUND'); + expect(result.current.auditResults).toBeNull(); + expect(mockRequest).toHaveBeenCalledWith( + GET_AUDIT_QUERY, + { auditId: 'eval-1' }, + { organization: 'org-1' } + ); + expect(mockRequest).not.toHaveBeenCalledWith( + GET_AUDIT_RESULTS_QUERY, + expect.anything(), + expect.anything() + ); + expect(mockRequest).not.toHaveBeenCalledWith( + GET_AUDIT_SUMMARY, + expect.anything(), + expect.anything() + ); + }); + + it('fetches summary and results when playground audit is completed', async () => { + mockRequest.mockImplementation((query: string) => { + if (query === GET_AUDIT_QUERY) { + return Promise.resolve({ + audit: makePlaygroundAudit({ + status: 'COMPLETED', + completedAt: '2026-01-02T00:00:00Z', + evaluationMode: 'manual', + }), + }); + } + if (query === GET_AUDIT_SUMMARY) { + return Promise.resolve({ + auditSummaries: [ + { + hasReport: false, + riskDistribution: null, + metricSummary: null, + recommendations: null, + auditReport: null, + }, + ], + }); + } + if (query === GET_AUDIT_RESULTS_QUERY) { + return Promise.resolve({ auditResults: [] }); + } + return Promise.resolve({}); + }); + + const { result } = renderHook(() => useAuditData('eval-1', 'org-1')); + + await waitFor(() => { + expect(result.current.auditResults).toEqual([]); + }); + + expect(mockRequest).toHaveBeenCalledWith( + GET_AUDIT_SUMMARY, + { audit_id: 'eval-1' }, + { organization: 'org-1' } + ); + expect(mockRequest).toHaveBeenCalledWith( + GET_AUDIT_RESULTS_QUERY, + { auditId: 'eval-1', metric: null }, + { organization: 'org-1' } + ); + }); +}); diff --git a/features/dashboard/components/EvaluationDetail/hooks/use-audit-polling.test.ts b/features/dashboard/components/EvaluationDetail/hooks/use-audit-polling.test.ts new file mode 100644 index 0000000..23b4d1f --- /dev/null +++ b/features/dashboard/components/EvaluationDetail/hooks/use-audit-polling.test.ts @@ -0,0 +1,133 @@ +import { act, renderHook } from '@testing-library/react'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import { GET_AUDIT_QUERY } from '@/features/dashboard/api/evaluation-queries'; +import { makePlaygroundAudit } from '@/testing/fixtures/playground-evaluation'; +import { useAuditPolling } from './use-audit-polling'; + +const mockRequest = vi.fn(); +const mockSetAudit = vi.fn(); +const mockSetEvaluationProgress = vi.fn(); +const mockFetchAuditSummary = vi.fn(); +const mockFetchAuditResults = vi.fn(); + +vi.mock('@/lib/graphql-client', () => ({ + useGraphQL: () => ({ request: mockRequest }), +})); + +describe('useAuditPolling playground behavior', () => { + beforeEach(() => { + vi.useFakeTimers(); + vi.clearAllMocks(); + mockFetchAuditSummary.mockResolvedValue({ hasReport: false }); + mockFetchAuditResults.mockResolvedValue(undefined); + }); + + afterEach(() => { + vi.useRealTimers(); + }); + + const renderPolling = (audit = makePlaygroundAudit({ status: 'IN_PROGRESS' })) => + renderHook( + ({ currentAudit }) => + useAuditPolling({ + evaluationId: 'eval-1', + orgId: 'org-1', + audit: currentAudit, + isAuthenticated: true, + isSessionLoading: false, + setAudit: mockSetAudit, + setEvaluationProgress: mockSetEvaluationProgress, + fetchAuditSummary: mockFetchAuditSummary, + fetchAuditResults: mockFetchAuditResults, + }), + { initialProps: { currentAudit: audit } } + ); + + it('continues polling while playground audit is IN_PROGRESS', async () => { + mockRequest.mockResolvedValue({ + audit: makePlaygroundAudit({ + status: 'IN_PROGRESS', + evaluationMode: 'PLAYGROUND', + progressPercentage: 40, + }), + }); + + renderPolling(); + + await act(async () => { + await vi.advanceTimersByTimeAsync(0); + }); + + expect(mockRequest).toHaveBeenCalledTimes(1); + expect(mockFetchAuditResults).not.toHaveBeenCalled(); + + await act(async () => { + await vi.advanceTimersByTimeAsync(5000); + }); + + expect(mockRequest).toHaveBeenCalledTimes(2); + }); + + it('stops polling and fetches results when playground audit completes', async () => { + mockRequest.mockResolvedValue({ + audit: makePlaygroundAudit({ + status: 'COMPLETED', + completedAt: '2026-01-02T00:00:00Z', + evaluationMode: 'manual', + }), + }); + + renderPolling(); + + await act(async () => { + await vi.runOnlyPendingTimersAsync(); + }); + + expect(mockRequest).toHaveBeenCalledWith( + GET_AUDIT_QUERY, + { auditId: 'eval-1' }, + { organization: 'org-1' } + ); + expect(mockFetchAuditSummary).toHaveBeenCalled(); + expect(mockFetchAuditResults).toHaveBeenCalled(); + + mockRequest.mockClear(); + + await act(async () => { + await vi.advanceTimersByTimeAsync(5000); + }); + + expect(mockRequest).not.toHaveBeenCalled(); + }); + + it('does not stop polling at PENDING_REVIEW for playground audits', async () => { + mockRequest.mockResolvedValue({ + audit: makePlaygroundAudit({ + status: 'PENDING_REVIEW', + evaluationMode: 'PLAYGROUND', + }), + }); + + renderPolling(); + + await act(async () => { + await vi.runOnlyPendingTimersAsync(); + }); + + expect(mockFetchAuditResults).not.toHaveBeenCalled(); + + mockRequest.mockClear(); + mockRequest.mockResolvedValue({ + audit: makePlaygroundAudit({ + status: 'PENDING_REVIEW', + evaluationMode: 'PLAYGROUND', + }), + }); + + await act(async () => { + await vi.advanceTimersByTimeAsync(5000); + }); + + expect(mockRequest).toHaveBeenCalledTimes(1); + }); +}); diff --git a/features/dashboard/components/EvaluationDetail/hooks/use-evaluation-detail.test.ts b/features/dashboard/components/EvaluationDetail/hooks/use-evaluation-detail.test.ts new file mode 100644 index 0000000..5e5e354 --- /dev/null +++ b/features/dashboard/components/EvaluationDetail/hooks/use-evaluation-detail.test.ts @@ -0,0 +1,133 @@ +import { renderHook, waitFor } from '@testing-library/react'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; +import { makePlaygroundAudit } from '@/testing/fixtures/playground-evaluation'; +import { useAuditActions } from './use-audit-actions'; +import { useAuditData } from './use-audit-data'; +import { useEvaluationDetail } from './use-evaluation-detail'; + +const mockStopProgressPolling = vi.fn(); +const mockSubmitBulkReview = vi.fn(); + +vi.mock('./use-audit-data', () => ({ + useAuditData: vi.fn(), +})); + +vi.mock('./use-audit-polling', () => ({ + useAuditPolling: vi.fn(() => ({ stopProgressPolling: mockStopProgressPolling })), +})); + +vi.mock('./use-audit-actions', () => ({ + useAuditActions: vi.fn(), +})); + +const mockedUseAuditData = vi.mocked(useAuditData); +const mockedUseAuditActions = vi.mocked(useAuditActions); + +function setupAuditData( + overrides: { + audit?: ReturnType | null; + auditReport?: { name: string; size: number | null; url: string } | null; + } = {} +) { + const audit = overrides.audit === undefined ? makePlaygroundAudit() : overrides.audit; + + mockedUseAuditData.mockReturnValue({ + audit, + auditResults: [], + auditReport: overrides.auditReport ?? null, + riskDistribution: { LOW_RISK: 0, MEDIUM_RISK: 0, HIGH_RISK: 0 }, + metricSummary: {}, + evaluatorRecommendation: '', + evaluationProgress: 25, + isLoading: false, + isSessionLoading: false, + isAuthenticated: true, + error: null, + setAudit: vi.fn(), + setEvaluatorRecommendation: vi.fn(), + setEvaluationProgress: vi.fn(), + fetchAuditSummary: vi.fn().mockResolvedValue({ hasReport: false }), + fetchAuditResults: vi.fn().mockResolvedValue(undefined), + } as ReturnType); +} + +function setupAuditActions() { + mockedUseAuditActions.mockReturnValue({ + isSavingName: false, + isSavingEvaluation: false, + isGeneratingReport: false, + isDownloading: false, + showSubmitRecommendationModal: false, + setShowSubmitRecommendationModal: vi.fn(), + saveEvaluationName: vi.fn(), + submitBulkReview: mockSubmitBulkReview, + generateReport: vi.fn(), + downloadReport: vi.fn(), + }); +} + +describe('useEvaluationDetail playground computed flags', () => { + beforeEach(() => { + vi.clearAllMocks(); + setupAuditActions(); + }); + + it('identifies playground evaluations', () => { + setupAuditData({ audit: makePlaygroundAudit({ evaluationMode: 'PLAYGROUND' }) }); + + const { result } = renderHook(() => useEvaluationDetail('eval-1', 'org-1')); + + expect(result.current.computed.isPlaygroundEvaluation).toBe(true); + expect(result.current.computed.isBulkPendingReview).toBe(false); + expect(result.current.computed.isBulkCompleted).toBe(false); + }); + + it('marks playground in progress when status is IN_PROGRESS', () => { + setupAuditData({ + audit: makePlaygroundAudit({ status: 'IN_PROGRESS', evaluationMode: 'manual' }), + }); + + const { result } = renderHook(() => useEvaluationDetail('eval-1', 'org-1')); + + expect(result.current.computed.isPlaygroundInProgress).toBe(true); + expect(result.current.computed.isRunning).toBe(true); + }); + + it('shows download actions only after playground completion', () => { + setupAuditData({ + audit: makePlaygroundAudit({ + status: 'COMPLETED', + completedAt: '2026-01-02T00:00:00Z', + }), + }); + + const { result } = renderHook(() => useEvaluationDetail('eval-1', 'org-1')); + + expect(result.current.computed.isEvaluationComplete).toBe(true); + expect(result.current.computed.showDownloadActions).toBe(true); + }); + + it('hides download actions while playground is in progress', () => { + setupAuditData({ + audit: makePlaygroundAudit({ status: 'IN_PROGRESS', completedAt: null }), + }); + + const { result } = renderHook(() => useEvaluationDetail('eval-1', 'org-1')); + + expect(result.current.computed.showDownloadActions).toBe(false); + }); + + it('shows report-ready state when audit report exists', () => { + setupAuditData({ + audit: makePlaygroundAudit({ + status: 'COMPLETED', + completedAt: '2026-01-02T00:00:00Z', + }), + auditReport: { name: 'report.pdf', size: 100, url: 'https://example.com/report.pdf' }, + }); + + const { result } = renderHook(() => useEvaluationDetail('eval-1', 'org-1')); + + expect(result.current.computed.isReportReady).toBe(true); + }); +}); diff --git a/features/dashboard/utils/evaluation.test.ts b/features/dashboard/utils/evaluation.test.ts new file mode 100644 index 0000000..83df098 --- /dev/null +++ b/features/dashboard/utils/evaluation.test.ts @@ -0,0 +1,67 @@ +import { describe, expect, it } from 'vitest'; +import { + canShowEvaluationResults, + getModeLabel, + isPlaygroundEvaluationMode, + shouldStopPolling, +} from './evaluation'; + +describe('isPlaygroundEvaluationMode', () => { + it('returns true for manual and playground modes', () => { + expect(isPlaygroundEvaluationMode('manual')).toBe(true); + expect(isPlaygroundEvaluationMode('playground')).toBe(true); + expect(isPlaygroundEvaluationMode('PLAYGROUND')).toBe(true); + }); + + it('returns false for bulk modes', () => { + expect(isPlaygroundEvaluationMode('bulk')).toBe(false); + expect(isPlaygroundEvaluationMode('BULK')).toBe(false); + expect(isPlaygroundEvaluationMode('automated')).toBe(false); + }); + + it('returns false for null and undefined', () => { + expect(isPlaygroundEvaluationMode(null)).toBe(false); + expect(isPlaygroundEvaluationMode(undefined)).toBe(false); + }); +}); + +describe('canShowEvaluationResults for playground', () => { + it('does not show results while in progress', () => { + expect(canShowEvaluationResults({ status: 'IN_PROGRESS', completedAt: null }, true)).toBe( + false + ); + expect(canShowEvaluationResults({ status: 'PENDING_REVIEW', completedAt: null }, true)).toBe( + false + ); + }); + + it('shows results when completed', () => { + expect(canShowEvaluationResults({ status: 'COMPLETED', completedAt: '2026-01-02' }, true)).toBe( + true + ); + expect( + canShowEvaluationResults({ status: 'IN_PROGRESS', completedAt: '2026-01-02' }, true) + ).toBe(true); + }); +}); + +describe('shouldStopPolling for playground', () => { + it('does not stop at PENDING_REVIEW', () => { + expect(shouldStopPolling({ status: 'PENDING_REVIEW', completedAt: null }, true)).toBe(false); + }); + + it('stops when completed', () => { + expect(shouldStopPolling({ status: 'COMPLETED', completedAt: '2026-01-02' }, true)).toBe(true); + }); + + it('continues while in progress', () => { + expect(shouldStopPolling({ status: 'IN_PROGRESS', completedAt: null }, true)).toBe(false); + }); +}); + +describe('getModeLabel for playground', () => { + it('returns Playground Evaluation for manual and playground modes', () => { + expect(getModeLabel('manual')).toBe('Playground Evaluation'); + expect(getModeLabel('playground')).toBe('Playground Evaluation'); + }); +}); diff --git a/package.json b/package.json index 6f93a17..7fbeb29 100644 --- a/package.json +++ b/package.json @@ -18,6 +18,9 @@ "prepare": "husky", "postinstall": "npm run generate:ci", "test": "vitest", + "test:run": "vitest run", + "test:bulk": "vitest run features/ai-maker/utils/map-audit-results.test.ts features/dashboard/components/AddIssueModal.test.tsx features/dashboard/components/AuditResultsList.test.tsx features/dashboard/components/BulkTestCaseDetailSheet.test.tsx features/dashboard/components/EvaluationDetail/hooks/use-audit-actions.test.ts", + "test:playground": "vitest run features/ai-maker/components/manual-evaluation features/dashboard/components/EvaluationDetail/hooks/use-audit-data.test.ts", "test:coverage": "vitest run --coverage" }, "dependencies": { diff --git a/testing/fixtures/bulk-evaluation/audit-results.ts b/testing/fixtures/bulk-evaluation/audit-results.ts new file mode 100644 index 0000000..79d99d8 --- /dev/null +++ b/testing/fixtures/bulk-evaluation/audit-results.ts @@ -0,0 +1,143 @@ +/** + * Bulk evaluation test fixtures. + * Conventions: import vitest helpers explicitly; use these factories for AuditResult shapes. + */ +import type { BulkTestCase } from '@/features/ai-maker/types/bulk-evaluation'; +import type { AuditResult } from '@/features/ai-maker/utils/map-audit-results'; + +let counter = 0; + +const nextId = (prefix: string) => { + counter += 1; + return `${prefix}-${counter}`; +}; + +export function resetFixtureCounter() { + counter = 0; +} + +export function makeAuditResult( + overrides: Partial & { testId?: string } = {} +): AuditResult { + const id = overrides.id ?? nextId('result'); + const testId = overrides.testId ?? nextId('test'); + const { testId: _testId, ...rest } = overrides; + + return { + id, + success: false, + riskLevel: 'HIGH', + reason: 'Test failure reason', + task: { + id: nextId('task'), + module: 'BIAS_FAIRNESS', + moduleDisplayName: 'Bias and Fairness', + metric: 'gender_bias', + metricDisplayName: 'Gender Bias', + test: { + id: testId, + testInput: 'What is the role of women in society?', + actualOutput: 'Women should stay at home.', + }, + }, + ...rest, + }; +} + +export function makePassingResult( + overrides: Partial & { testId?: string } = {} +): AuditResult { + return makeAuditResult({ + success: true, + riskLevel: 'NO_RISK', + reason: null, + ...overrides, + }); +} + +export function makeReviewedResult( + overrides: Partial & { testId?: string } = {} +): AuditResult { + return makeAuditResult({ + isReviewed: true, + evaluatorSuccess: false, + evaluatorRiskLevel: 'HIGH', + evaluatorReason: 'Reviewer noted bias', + ...overrides, + }); +} + +export function makeReviewedPassingResult( + overrides: Partial & { testId?: string } = {} +): AuditResult { + return makeAuditResult({ + isReviewed: true, + evaluatorSuccess: true, + evaluatorRiskLevel: 'NO_RISK', + ...overrides, + }); +} + +/** Two results sharing the same test id (multi-metric on one input). */ +export function makeGroupedAuditResults(testId = 'shared-test-1'): AuditResult[] { + return [ + makeAuditResult({ + testId, + id: 'result-metric-a', + riskLevel: 'HIGH', + reason: 'Bias detected', + task: { + id: 'task-a', + module: 'BIAS_FAIRNESS', + moduleDisplayName: 'Bias and Fairness', + metric: 'gender_bias', + metricDisplayName: 'Gender Bias', + test: { + id: testId, + testInput: 'Line one\nLine two prompt', + actualOutput: 'Output text', + }, + }, + }), + makeAuditResult({ + testId, + id: 'result-metric-b', + riskLevel: 'MEDIUM', + reason: 'Safety concern', + task: { + id: 'task-b', + module: 'PRIVACY_SAFETY', + moduleDisplayName: 'Privacy and Safety', + metric: 'safety_risk', + metricDisplayName: 'Safety Risk', + test: { + id: testId, + testInput: 'Line one\nLine two prompt', + actualOutput: 'Output text', + }, + }, + }), + ]; +} + +export function makeBulkTestCase(overrides: Partial = {}): BulkTestCase { + return { + id: 'test-case-1', + index: 1, + moduleId: 'BIAS_FAIRNESS', + moduleDisplayName: 'Bias and Fairness', + inputPrompt: 'What is the role of women in society?', + fullInputText: 'What is the role of women in society?', + output: 'Women should stay at home.', + risks: [ + { + resultId: 'result-1', + severity: 'HIGH', + label: 'Gender Bias', + observation: 'Biased response detected', + }, + ], + allMetricResults: [{ resultId: 'result-1', label: 'Gender Bias', metricKey: 'gender_bias' }], + ...overrides, + }; +} diff --git a/testing/fixtures/bulk-evaluation/audit.ts b/testing/fixtures/bulk-evaluation/audit.ts new file mode 100644 index 0000000..bf6df73 --- /dev/null +++ b/testing/fixtures/bulk-evaluation/audit.ts @@ -0,0 +1,26 @@ +import type { Audit } from '@/features/dashboard/types/audit'; + +export function makeBulkAudit(overrides: Partial = {}): Audit { + return { + auditType: 'TECHNICAL', + evaluationMode: 'BULK', + id: 'audit-1', + name: 'Bulk Evaluation 1', + modelId: 'model-1', + modelName: 'Test Model', + status: 'PENDING_REVIEW', + modules: ['BIAS_FAIRNESS'], + metrics: ['gender_bias'], + configuration: {}, + totalTests: 10, + passedTests: 8, + failedTests: 2, + skippedTests: 0, + errorMessage: null, + errorDetails: null, + createdAt: '2026-01-01T00:00:00Z', + startedAt: '2026-01-01T00:00:00Z', + completedAt: null, + ...overrides, + }; +} diff --git a/testing/fixtures/bulk-evaluation/index.ts b/testing/fixtures/bulk-evaluation/index.ts new file mode 100644 index 0000000..8e9c170 --- /dev/null +++ b/testing/fixtures/bulk-evaluation/index.ts @@ -0,0 +1,2 @@ +export * from './audit-results'; +export * from './audit'; diff --git a/testing/fixtures/playground-evaluation/audit.ts b/testing/fixtures/playground-evaluation/audit.ts new file mode 100644 index 0000000..91b6090 --- /dev/null +++ b/testing/fixtures/playground-evaluation/audit.ts @@ -0,0 +1,27 @@ +import type { Audit } from '@/features/dashboard/types/audit'; + +export function makePlaygroundAudit(overrides: Partial = {}): Audit { + return { + auditType: 'TECHNICAL', + evaluationMode: 'PLAYGROUND', + id: 'playground-audit-1', + name: 'Playground Evaluation 1', + modelId: 'model-1', + modelName: 'Test Model', + status: 'IN_PROGRESS', + modules: ['BIAS_FAIRNESS'], + metrics: ['GENDER_BIAS'], + configuration: {}, + totalTests: 0, + passedTests: 0, + failedTests: 0, + skippedTests: 0, + errorMessage: null, + errorDetails: null, + createdAt: '2026-01-01T00:00:00Z', + startedAt: '2026-01-01T00:00:00Z', + completedAt: null, + progressPercentage: 0, + ...overrides, + }; +} diff --git a/testing/fixtures/playground-evaluation/index.ts b/testing/fixtures/playground-evaluation/index.ts new file mode 100644 index 0000000..f4441e5 --- /dev/null +++ b/testing/fixtures/playground-evaluation/index.ts @@ -0,0 +1,2 @@ +export * from './audit'; +export * from './manual-test-cases'; diff --git a/testing/fixtures/playground-evaluation/manual-test-cases.ts b/testing/fixtures/playground-evaluation/manual-test-cases.ts new file mode 100644 index 0000000..70a6449 --- /dev/null +++ b/testing/fixtures/playground-evaluation/manual-test-cases.ts @@ -0,0 +1,86 @@ +import type { + ManualEvalWorkspaceDraft, + ManualTestCase, + ManualTestCaseIssue, + ModuleProgress, + PlaygroundEvaluationStatus, +} from '@/features/ai-maker/components/manual-evaluation/types'; + +export function makeManualTestCaseIssue( + overrides: Partial = {} +): ManualTestCaseIssue { + return { + metricName: 'GENDER_BIAS', + status: 'FAILED', + severity: 'HIGH', + comments: 'Biased output detected', + idealOutput: '', + ...overrides, + }; +} + +export function makeManualTestCase(overrides: Partial = {}): ManualTestCase { + return { + id: 'manual-test-1', + testInput: 'What is the role of women in society?', + actualOutput: 'Women should stay at home.', + issues: [makeManualTestCaseIssue()], + createdAt: '2026-01-01T10:00:00Z', + ...overrides, + }; +} + +export function makePassingManualTestCase(overrides: Partial = {}): ManualTestCase { + return makeManualTestCase({ + issues: [{ metricName: 'GENDER_BIAS', status: 'PASSED' }], + ...overrides, + }); +} + +export function makeModuleProgress(overrides: Partial = {}): ModuleProgress { + return { + module: 'BIAS_FAIRNESS', + moduleDisplayName: 'Bias and Fairness', + testCaseCount: 2, + isComplete: false, + canComplete: false, + passedCount: 1, + failedCount: 1, + ...overrides, + }; +} + +export function makePlaygroundStatus( + overrides: Partial = {} +): PlaygroundEvaluationStatus { + return { + auditId: 'playground-audit-1', + testCaseCount: 2, + canFinish: false, + ...overrides, + }; +} + +export function makeWorkspaceDraft( + overrides: Partial = {} +): ManualEvalWorkspaceDraft { + return { + selectedModule: 'BIAS_FAIRNESS', + sourceLanguage: 'en', + targetLanguage: 'hi', + inputPrompt: 'Test prompt', + modelOutput: 'Model response', + hasCalledModel: true, + status: null, + issueRows: [ + { + id: 'issue-row-1', + issueType: 'GENDER_BIAS', + severity: 'HIGH', + observations: 'Bias noted', + idealOutput: '', + }, + ], + ...overrides, + }; +} diff --git a/testing/mocks/graphql-client.ts b/testing/mocks/graphql-client.ts new file mode 100644 index 0000000..5a33aca --- /dev/null +++ b/testing/mocks/graphql-client.ts @@ -0,0 +1,5 @@ +import { vi } from 'vitest'; + +export function createMockGraphQLRequest() { + return vi.fn(); +} diff --git a/testing/mocks/index.ts b/testing/mocks/index.ts new file mode 100644 index 0000000..fc7b6d5 --- /dev/null +++ b/testing/mocks/index.ts @@ -0,0 +1,2 @@ +export * from './graphql-client'; +export { getOpubUiMockModule, mockToast, opubUiMock } from './opub-ui'; diff --git a/testing/mocks/opub-ui.tsx b/testing/mocks/opub-ui.tsx new file mode 100644 index 0000000..110dee2 --- /dev/null +++ b/testing/mocks/opub-ui.tsx @@ -0,0 +1,224 @@ +import React from 'react'; +import { vi } from 'vitest'; + +export const mockToast = { + success: vi.fn(), + error: vi.fn(), +}; + +type SelectProps = { + name: string; + label?: string; + options?: Array<{ value: string; label: string }>; + value?: string; + onChange?: (value: string) => void; + error?: string; + disabled?: boolean; + placeholder?: string; +}; + +function SelectMock({ + name, + label, + options = [], + value = '', + onChange, + error, + disabled, +}: SelectProps) { + return ( +
+ {label && } + + {error && {error}} +
+ ); +} + +type ComboboxProps = { + name: string; + label?: string; + list?: Array<{ value: string; label: string }>; + selectedValue?: string; + onChange?: (value: string | Array<{ value: string; label: string }>) => void; +}; + +function ComboboxMock({ name, label, list = [], selectedValue = '', onChange }: ComboboxProps) { + const currentValue = + list.find((item) => item.label === selectedValue || item.value === selectedValue)?.value ?? + selectedValue; + + return ( +
+ {label && } + +
+ ); +} + +type TextFieldProps = { + name: string; + label?: string; + value?: string; + onChange?: (value: string) => void; + error?: string; + readOnly?: boolean; + multiline?: number; +}; + +function TextFieldMock({ + name, + label, + value = '', + onChange, + error, + readOnly, + multiline, +}: TextFieldProps) { + const common = { + id: name, + name, + 'aria-label': label || name, + value, + readOnly, + onChange: (e: React.ChangeEvent) => + onChange?.(e.target.value), + }; + + return ( +
+ {label && } + {multiline ?