diff --git a/README.md b/README.md
index 8d9f2744..3cdd0b5e 100644
--- a/README.md
+++ b/README.md
@@ -14,7 +14,7 @@
-
+
@@ -510,7 +510,7 @@ Operator notes for activating existing adapters, metasearch landings on the dire | OTA Channels | Booking.com + Expedia (EQC) + SiteMinder + DerbySoft | Direct + aggregated OTA connectivity (ARI + content) | | XML Processing | fast-xml-parser | Booking.com OTA XML protocol | | Package Manager | pnpm workspaces | Monorepo management | -| Testing | Vitest (2392 passing tests across 280 files with passing tests) | Unit and integration tests | +| Testing | Vitest (2407 passing tests across 281 files with passing tests) | Unit and integration tests | | Build | tsup (packages) + Vite (dashboard) + nest build (API) | Fast builds | | Containers | Docker + docker-compose | Local dev and production deployment | | CI/CD | GitHub Actions | Automated testing, builds, and releases | @@ -648,7 +648,7 @@ Before going live, verify the items in [`docs/deployment.md`](./docs/deployment. ### Run tests ```bash -# Passing-test count: 2392 test cases across 280 files (skipped excluded) +# Passing-test count: 2407 test cases across 281 files (skipped excluded) # API tests only pnpm --filter @telivityhaip/api test @@ -1197,7 +1197,7 @@ HAIP is built in public and contributions are welcome. pnpm install # Install dependencies pnpm build # Build all workspace packages pnpm dev # Start API in dev mode (hot reload) -pnpm test # Run all tests (2392 passing, 280 files with passes; skipped excluded) +pnpm test # Run all tests (2407 passing, 281 files with passes; skipped excluded) pnpm lint # ESLint ``` diff --git a/apps/api/src/modules/agent/agent-autopilot-tiers.spec.ts b/apps/api/src/modules/agent/agent-autopilot-tiers.spec.ts new file mode 100644 index 00000000..54792d5f --- /dev/null +++ b/apps/api/src/modules/agent/agent-autopilot-tiers.spec.ts @@ -0,0 +1,119 @@ +import { describe, it, expect } from 'vitest'; +import { + autopilotRiskTier, + shouldAutoExecuteDecision, + MONEY_AUTOPILOT_FLOOR, +} from './agent-autopilot-tiers'; + +describe('autopilotRiskTier', () => { + it('classifies money-moving agents', () => { + expect(autopilotRiskTier('pricing')).toBe('money'); + expect(autopilotRiskTier('overbooking')).toBe('money'); + expect(autopilotRiskTier('channel_mix')).toBe('money'); + expect(autopilotRiskTier('revenue_manager')).toBe('money'); + expect(autopilotRiskTier('group_pickup')).toBe('money'); + }); + + it('classifies guest-facing agents', () => { + expect(autopilotRiskTier('guest_comms')).toBe('guest'); + expect(autopilotRiskTier('review_response')).toBe('guest'); + }); + + it('classifies remaining agents as ops', () => { + expect(autopilotRiskTier('demand_forecast')).toBe('ops'); + expect(autopilotRiskTier('night_audit')).toBe('ops'); + expect(autopilotRiskTier('housekeeping')).toBe('ops'); + expect(autopilotRiskTier('cancellation')).toBe('ops'); + expect(autopilotRiskTier('ar_collections')).toBe('ops'); + expect(autopilotRiskTier('deposit_risk')).toBe('ops'); + }); +}); + +describe('shouldAutoExecuteDecision', () => { + it('never auto-executes outside autopilot mode', () => { + expect( + shouldAutoExecuteDecision({ + mode: 'suggest', + agentType: 'pricing', + confidence: 0.99, + configThreshold: 0.85, + }), + ).toBe(false); + }); + + it('never auto-executes guest-facing agents even at high confidence', () => { + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'review_response', + confidence: 0.99, + configThreshold: 0.5, + }), + ).toBe(false); + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'guest_comms', + confidence: 0.99, + configThreshold: 0.5, + }), + ).toBe(false); + }); + + it('requires money agents to clear the money floor even if config is lower', () => { + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'pricing', + confidence: 0.9, + configThreshold: 0.85, + }), + ).toBe(false); + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'pricing', + confidence: MONEY_AUTOPILOT_FLOOR, + configThreshold: 0.85, + }), + ).toBe(true); + }); + + it('respects a config threshold above the money floor', () => { + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'overbooking', + confidence: 0.93, + configThreshold: 0.95, + }), + ).toBe(false); + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'overbooking', + confidence: 0.95, + configThreshold: 0.95, + }), + ).toBe(true); + }); + + it('uses the config threshold alone for ops agents', () => { + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'demand_forecast', + confidence: 0.85, + configThreshold: 0.85, + }), + ).toBe(true); + expect( + shouldAutoExecuteDecision({ + mode: 'autopilot', + agentType: 'night_audit', + confidence: 0.84, + configThreshold: 0.85, + }), + ).toBe(false); + }); +}); diff --git a/apps/api/src/modules/agent/agent-autopilot-tiers.ts b/apps/api/src/modules/agent/agent-autopilot-tiers.ts new file mode 100644 index 00000000..1ed9bfd8 --- /dev/null +++ b/apps/api/src/modules/agent/agent-autopilot-tiers.ts @@ -0,0 +1,50 @@ +/** + * Autopilot risk tiers — one confidence number is not equal across agents. + * Money-moving writes need a harder floor; guest-facing drafts never auto-run. + */ + +export type AutopilotRiskTier = 'money' | 'ops' | 'guest'; + +const MONEY_AGENTS = new Set([ + 'pricing', + 'overbooking', + 'channel_mix', + 'revenue_manager', + 'group_pickup', +]); + +const GUEST_AGENTS = new Set(['guest_comms', 'review_response']); + +/** Floor applied on top of the property's configured autopilot threshold for money agents. */ +export const MONEY_AUTOPILOT_FLOOR = 0.92; + +export function autopilotRiskTier(agentType: string): AutopilotRiskTier { + if (MONEY_AGENTS.has(agentType)) return 'money'; + if (GUEST_AGENTS.has(agentType)) return 'guest'; + return 'ops'; +} + +/** + * Whether autopilot may auto-execute this recommendation. + * Guest-facing agents always return false (human approve). + * Money agents require confidence >= max(configThreshold, MONEY_AUTOPILOT_FLOOR). + * Ops agents use the config threshold as today. + */ +export function shouldAutoExecuteDecision(input: { + mode: string; + agentType: string; + confidence: number; + configThreshold: number; +}): boolean { + if (input.mode !== 'autopilot') return false; + + const tier = autopilotRiskTier(input.agentType); + if (tier === 'guest') return false; + + const threshold = + tier === 'money' + ? Math.max(input.configThreshold, MONEY_AUTOPILOT_FLOOR) + : input.configThreshold; + + return input.confidence >= threshold; +} diff --git a/apps/api/src/modules/agent/agent.service.spec.ts b/apps/api/src/modules/agent/agent.service.spec.ts index cadae08f..0cb4e335 100644 --- a/apps/api/src/modules/agent/agent.service.spec.ts +++ b/apps/api/src/modules/agent/agent.service.spec.ts @@ -238,6 +238,139 @@ describe('AgentService', () => { ); }); + it('does not auto-execute money agents below the 0.92 floor in autopilot', async () => { + const autopilotConfig = { + ...existingConfig, + mode: 'autopilot', + autopilotConfidenceThreshold: '0.85', + }; + const insertedDecision = { + id: 'dec-money', + decisionType: 'rate_adjustment', + confidence: '0.90', + status: 'pending', + }; + const db = createMockDb({ + selectResult: [autopilotConfig], + insertResult: [insertedDecision], + }); + const execute = vi.fn().mockResolvedValue({ success: true, changes: [] }); + const agent = createMockAgent('pricing'); + agent.recommend = async () => [ + { + decisionType: 'rate_adjustment', + recommendation: { delta: 10 }, + confidence: 0.9, + inputSnapshot: {}, + }, + ]; + agent.execute = execute; + + const module = await Test.createTestingModule({ + providers: [ + AgentService, + { provide: DRIZZLE, useValue: db }, + { provide: WebhookService, useValue: { emit: vi.fn().mockResolvedValue(undefined) } }, + { provide: LlmService, useValue: { explain: vi.fn().mockResolvedValue(null) } }, + ], + }).compile(); + const service = module.get(AgentService); + service.registerAgent(agent); + + const result = await service.runAgent('prop-1', 'pricing'); + expect(execute).not.toHaveBeenCalled(); + expect((result as any).decisions[0].status).toBe('pending'); + }); + + it('never auto-executes review_response even at high confidence in autopilot', async () => { + const autopilotConfig = { + ...existingConfig, + agentType: 'review_response', + mode: 'autopilot', + autopilotConfidenceThreshold: '0.50', + }; + const insertedDecision = { + id: 'dec-review', + decisionType: 'review_response', + confidence: '0.99', + status: 'pending', + }; + const db = createMockDb({ + selectResult: [autopilotConfig], + insertResult: [insertedDecision], + }); + const execute = vi.fn().mockResolvedValue({ success: true, changes: [] }); + const agent = createMockAgent('review_response'); + agent.recommend = async () => [ + { + decisionType: 'review_response', + recommendation: { responseText: 'Thanks' }, + confidence: 0.99, + inputSnapshot: {}, + }, + ]; + agent.execute = execute; + + const module = await Test.createTestingModule({ + providers: [ + AgentService, + { provide: DRIZZLE, useValue: db }, + { provide: WebhookService, useValue: { emit: vi.fn().mockResolvedValue(undefined) } }, + { provide: LlmService, useValue: { explain: vi.fn().mockResolvedValue(null) } }, + ], + }).compile(); + const service = module.get(AgentService); + service.registerAgent(agent); + + await service.runAgent('prop-1', 'review_response'); + expect(execute).not.toHaveBeenCalled(); + }); + + it('auto-executes ops agents at the configured threshold in autopilot', async () => { + const autopilotConfig = { + ...existingConfig, + agentType: 'demand_forecast', + mode: 'autopilot', + autopilotConfidenceThreshold: '0.85', + }; + const insertedDecision = { + id: 'dec-ops', + decisionType: 'forecast', + confidence: '0.85', + status: 'pending', + }; + const db = createMockDb({ + selectResult: [autopilotConfig], + insertResult: [insertedDecision], + updateResult: [{ ...insertedDecision, status: 'auto_executed' }], + }); + const execute = vi.fn().mockResolvedValue({ success: true, changes: [] }); + const agent = createMockAgent('demand_forecast'); + agent.recommend = async () => [ + { + decisionType: 'forecast', + recommendation: { summary: 'ok' }, + confidence: 0.85, + inputSnapshot: {}, + }, + ]; + agent.execute = execute; + + const module = await Test.createTestingModule({ + providers: [ + AgentService, + { provide: DRIZZLE, useValue: db }, + { provide: WebhookService, useValue: { emit: vi.fn().mockResolvedValue(undefined) } }, + { provide: LlmService, useValue: { explain: vi.fn().mockResolvedValue(null) } }, + ], + }).compile(); + const service = module.get(AgentService); + service.registerAgent(agent); + + await service.runAgent('prop-1', 'demand_forecast'); + expect(execute).toHaveBeenCalledOnce(); + }); + // --- approveDecision --- it('rejects approval of non-pending decision', async () => { diff --git a/apps/api/src/modules/agent/agent.service.ts b/apps/api/src/modules/agent/agent.service.ts index 5471d62f..680b01e5 100644 --- a/apps/api/src/modules/agent/agent.service.ts +++ b/apps/api/src/modules/agent/agent.service.ts @@ -19,6 +19,7 @@ import { isValidAgentType, subgraphFor, } from './agent-graph'; +import { shouldAutoExecuteDecision } from './agent-autopilot-tiers'; @Injectable() export class AgentService { @@ -119,8 +120,13 @@ export class AgentService { const threshold = parseFloat(config.autopilotConfidenceThreshold ?? '0.85'); for (const rec of recommendations) { - const shouldAutoExecute = - config.mode === 'autopilot' && rec.confidence >= threshold; + // Risk-tiered autopilot: money needs a harder floor; guest drafts never auto-run. + const shouldAutoExecute = shouldAutoExecuteDecision({ + mode: config.mode, + agentType, + confidence: rec.confidence, + configThreshold: threshold, + }); // Always insert as pending first — update to auto_executed only on success const [decision] = await this.db diff --git a/apps/api/src/modules/llm/grounding.spec.ts b/apps/api/src/modules/llm/grounding.spec.ts index b68460b9..0cb6eb82 100644 --- a/apps/api/src/modules/llm/grounding.spec.ts +++ b/apps/api/src/modules/llm/grounding.spec.ts @@ -103,6 +103,46 @@ describe('grounding — anti-hallucination guard', () => { expect(out.grounded).toBe(false); }); + it('treats $-amounts as significant even when under 25', () => { + const nums = significantNumbers('Fee is only $12'); + expect(nums).toContain(12); + const out = groundExplanation( + { fee: 5 }, + { rationale: 'Charge a $12 resort fee.', suggestions: [] }, + ); + expect(out.grounded).toBe(false); + }); + + it('allows signed phrasing when the magnitude is in the decision (sign via words)', () => { + // Documented: natural language carries sign ("cut", "reduce"); bare "-" is not required. + const out = groundExplanation( + { recommendedAdjustmentPct: 12 }, + { rationale: 'Cut the rate by 12%.', suggestions: [] }, + ); + expect(out.grounded).toBe(true); + }); + + it('does not let occupancy 100 support a hallucinated 1% via inverse scaling', () => { + expect(isSupported(1, new Set([100]))).toBe(false); + const out = groundExplanation( + { rooms: 100 }, + { rationale: 'Lift rates by 1%.', suggestions: [] }, + ); + expect(out.grounded).toBe(false); + }); + + it('drops suggestions that invent percentages while keeping grounded ones', () => { + const out = groundExplanation( + { occupancy: 0.9, recommendedAdjustmentPct: 8 }, + { + rationale: 'Occupancy is 90%.', + suggestions: ['Raise 8%', 'Also consider a secret 22% promo'], + }, + ); + expect(out.grounded).toBe(true); + expect(out.suggestions).toEqual(['Raise 8%']); + }); + // --- numericPayload: structural "numbers only" enforcement (finding #4) --- it('numericPayload keeps numeric leaves and drops all free-form strings', () => { const out = numericPayload({ diff --git a/docs/test-stats.json b/docs/test-stats.json index 5e182da4..8c7ade4d 100644 --- a/docs/test-stats.json +++ b/docs/test-stats.json @@ -1,7 +1,7 @@ { - "tests": 2392, - "files": 280, + "tests": 2407, + "files": 281, "scope": "all workspace packages with a test script", "semantics": "passed test cases and files containing at least one passed test; skipped test cases and skipped-only files are excluded", - "updatedAt": "2026-09-09T23:52:25.972Z" + "updatedAt": "2026-09-24T06:33:10.149Z" } diff --git a/specs/4-2-property-deduplication.yaml b/specs/4-2-property-deduplication.yaml index 949b85d4..87dd187b 100644 --- a/specs/4-2-property-deduplication.yaml +++ b/specs/4-2-property-deduplication.yaml @@ -2,14 +2,16 @@ agent: id: "4.2" name: Property Deduplication Agent domain: lodging - version: "0.1.0" + version: "0.2.0" status: live package: "@otaip/agents-lodging" description: > Identifies duplicate hotel properties across multiple sources and merges them - into canonical records. Uses a multi-algorithm scoring pipeline with configurable - thresholds and Union-Find grouping for transitive matches. + into canonical records. Cheap string/geo similarity only builds a shortlist of + candidate pairs. The final decision is a three-way outcome — same, maybe + (human review), or different — not a single magic similarity cutoff as merge + authority. Union-Find still groups transitive same matches. input: type: PropertyDeduplicationInput @@ -18,10 +20,10 @@ input: type: RawHotelResult[] required: true description: Raw hotel results from multiple sources (output of Agent 4.1) - - name: thresholds - type: DeduplicationThresholds + - name: shortlist + type: DeduplicationShortlistConfig required: false - description: Override default merge/review/separate thresholds + description: Override blocking / candidate-generation similarity settings - name: source_trust_hierarchy type: string[] required: false @@ -32,13 +34,18 @@ output: fields: - name: canonical_properties type: CanonicalProperty[] - description: Deduplicated properties with merged content + description: Deduplicated properties with merged content (same outcomes only) - name: merge_log type: MergeRecord[] - description: Which source records were merged and why (confidence + reasoning) + description: Which source records were merged as same and why - name: flagged_for_review type: PropertyPair[] - description: Pairs scoring between review and auto-merge thresholds + description: > + Candidate pairs judged maybe — human must decide. Include which fields + disagree (name, address, geo, chain, stars) to help the curator. + - name: separated_pairs + type: PropertyPair[] + description: Candidate pairs judged different (left unlinked) - name: total_input type: number - name: total_output @@ -48,13 +55,15 @@ output: description: Reduction percentage (e.g., 0.4 means 40% were duplicates) behavior: - - "Pipeline: Normalize → Block → Score → Threshold → Merge" - - "Scoring algorithms: Jaro-Winkler (name, 0.3), Levenshtein (address, 0.2), Haversine (coordinates, 0.25), chain code (0.15), star rating (0.1)" - - "Thresholds: >0.85 auto-merge, 0.65-0.85 flag for review, <0.65 separate" - - Blocking by city + chain code to reduce O(n²) comparisons - - Union-Find grouping for transitive matches (A matches B, B matches C → A,B,C merge) - - Content merge uses source trust hierarchy for per-field best-content selection - - Explainable — every merge includes confidence score and algorithm breakdown + - "Pipeline: Normalize → Block → Shortlist → Decide (same|maybe|different) → Merge same only" + - "Shortlist (not final authority): Jaro-Winkler (name), Levenshtein (address), Haversine (coordinates), chain code, star rating — weighted composite only proposes candidate pairs" + - "Blocking by city + chain code to reduce O(n²) comparisons before shortlist" + - "Final outcome per candidate pair is three-way: same (merge) | maybe (human queue + field disagreements) | different (leave apart)" + - "Do NOT auto-merge solely because a string/geo composite exceeds 0.85 — similarity scores propose candidates; semantic sameness decides" + - "When implemented, the intended final judge is a scored same/maybe/different judgment plus per-field agreement checks (name, address/geo, chain); string/geo scores remain shortlist-only" + - "Union-Find grouping only over pairs decided same (A same B, B same C → A,B,C merge)" + - "Content merge uses source trust hierarchy for per-field best-content selection" + - "Explainable — every same/maybe/different decision records shortlist scores and field-level agreement notes" errors: - AgentNotInitializedError: Called before initialize() diff --git a/specs/domain-4-lodging-agents.md b/specs/domain-4-lodging-agents.md index 0a58645b..95cb01cd 100644 --- a/specs/domain-4-lodging-agents.md +++ b/specs/domain-4-lodging-agents.md @@ -62,27 +62,27 @@ **Inputs:** - Raw hotel results from Agent 4.1 (multiple sources, unmerged) -- Matching configuration (thresholds, algorithm weights) +- Shortlist / blocking configuration (similarity only proposes candidates) - Source trust hierarchy (which source has best photos, most accurate coordinates, etc.) **Outputs:** -- Canonical property records (one per physical property) +- Canonical property records (one per physical property) for **same** decisions - Per-property: merged content (best name, best address, best coordinates, best photos, merged amenities) -- Match confidence score per merge decision +- Per-pair outcome: same / maybe / different, with field-level disagreement notes on maybe - Source attribution (which data came from which source) -- Unmatched properties (couldn't confidently merge — flagged for review) +- Maybe pairs flagged for human review (not auto-merged) **Matching pipeline:** 1. **Normalize** — standardize address components, strip noise words ("Hotel", "The", "Resort & Spa"), normalize chain names 2. **Block** — group candidates by coarse criteria (city + chain code) to reduce O(n²) comparison space -3. **Score** — multi-algorithm scoring: +3. **Shortlist** — multi-algorithm similarity scoring (proposes candidates only): - Jaro-Winkler on property name (weight: 0.3) - Levenshtein on normalized address (weight: 0.2) - Haversine distance on coordinates, 250m threshold (weight: 0.25) - Chain code exact match (weight: 0.15) - Star rating match (weight: 0.1) -4. **Threshold** — composite score > 0.85 = auto-merge, 0.65-0.85 = flag for review, < 0.65 = separate properties -5. **Merge** — combine best content per attribute using source trust hierarchy +4. **Decide** — for each shortlisted pair, outcome is **same** (merge) / **maybe** (human review, with which fields disagree) / **different** (leave apart). Cheap similarity scores only propose candidates; they are **not** merge authority. Do not auto-merge solely because a composite exceeds 0.85. +5. **Merge** — combine best content per attribute using source trust hierarchy — **same** pairs only **Content merge hierarchy (default):** 1. Hotel direct / chain CRS (most authoritative)