jackmichael commited on
Commit
7ba65c9
·
0 Parent(s):

Initial commit: RFP Engineering Platform with Deep Research, Inngest integration, and Cloud-ready configurations

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitignore +43 -0
  2. README.md +213 -0
  3. app/compliance/page.tsx +5 -0
  4. app/deep-research/page.tsx +5 -0
  5. app/editor/page.tsx +5 -0
  6. app/globals.css +86 -0
  7. app/ingestion/page.tsx +5 -0
  8. app/layout.tsx +41 -0
  9. app/opportunities/page.tsx +5 -0
  10. app/page.tsx +5 -0
  11. app/pricing/page.tsx +5 -0
  12. app/projects/page.tsx +5 -0
  13. app/settings/page.tsx +5 -0
  14. app/submission/page.tsx +5 -0
  15. app/vectors/page.tsx +5 -0
  16. backend/.env.example +70 -0
  17. backend/README.md +68 -0
  18. backend/ai_router.py +83 -0
  19. backend/app/__init__.py +1 -0
  20. backend/app/__pycache__/__init__.cpython-313.pyc +0 -0
  21. backend/app/__pycache__/database.cpython-313.pyc +0 -0
  22. backend/app/__pycache__/main.cpython-313.pyc +0 -0
  23. backend/app/__pycache__/models.cpython-313.pyc +0 -0
  24. backend/app/__pycache__/tasks.cpython-313.pyc +0 -0
  25. backend/app/ai/__init__.py +54 -0
  26. backend/app/ai/__pycache__/__init__.cpython-313.pyc +0 -0
  27. backend/app/ai/__pycache__/providers.cpython-313.pyc +0 -0
  28. backend/app/ai/__pycache__/smart_router.cpython-313.pyc +0 -0
  29. backend/app/ai/providers.py +533 -0
  30. backend/app/ai/smart_router.py +1097 -0
  31. backend/app/analysis/__init__.py +18 -0
  32. backend/app/analysis/__pycache__/__init__.cpython-313.pyc +0 -0
  33. backend/app/analysis/__pycache__/competitor_db.cpython-313.pyc +0 -0
  34. backend/app/analysis/__pycache__/models.cpython-313.pyc +0 -0
  35. backend/app/analysis/__pycache__/nli_detector.cpython-313.pyc +0 -0
  36. backend/app/analysis/__pycache__/service.cpython-313.pyc +0 -0
  37. backend/app/analysis/competitor_db.py +502 -0
  38. backend/app/analysis/models.py +66 -0
  39. backend/app/analysis/nli_detector.py +566 -0
  40. backend/app/analysis/service.py +922 -0
  41. backend/app/assembly/__init__.py +2 -0
  42. backend/app/assembly/__pycache__/__init__.cpython-313.pyc +0 -0
  43. backend/app/assembly/__pycache__/service.cpython-313.pyc +0 -0
  44. backend/app/assembly/service.py +939 -0
  45. backend/app/core/__init__.py +2 -0
  46. backend/app/core/__pycache__/__init__.cpython-313.pyc +0 -0
  47. backend/app/core/__pycache__/config.cpython-313.pyc +0 -0
  48. backend/app/core/config.py +59 -0
  49. backend/app/database.py +32 -0
  50. backend/app/ingestion/__init__.py +2 -0
.gitignore ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # dependencies
2
+ /node_modules
3
+ /.pnp
4
+ .pnp.js
5
+ .vercel
6
+
7
+ # testing
8
+ /coverage
9
+
10
+ # next.js
11
+ /.next/
12
+ /out/
13
+
14
+ # production
15
+ /build
16
+
17
+ # misc
18
+ .DS_Store
19
+ *.pem
20
+
21
+ # debug
22
+ npm-debug.log*
23
+ yarn-debug.log*
24
+ yarn-error.log*
25
+
26
+ # local env files
27
+ .env*.local
28
+
29
+ # vercel
30
+ .vercel
31
+
32
+ # typescript
33
+ *.tsbuildinfo
34
+ next-env.d.ts
35
+
36
+ # backend
37
+ backend/venv/
38
+ backend/.env
39
+ backend/__pycache__/
40
+ backend/chroma_db/
41
+ backend/uploads/
42
+ backend/outputs/
43
+ backend/*.log
README.md ADDED
@@ -0,0 +1,213 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Next-Generation RFP Automation Platform
2
+
3
+ An end-to-end enterprise proposal automation platform that transforms manual RFP response workflows into an intelligent, AI-driven pipeline.
4
+
5
+ ## Architecture Overview
6
+
7
+ The platform consists of six core engines:
8
+
9
+ | Engine | Purpose | Key Technologies |
10
+ | ------------- | ------------------------------------------- | ------------------------------ |
11
+ | **Ingestion** | PDF/DOCX parsing, OCR, table extraction | PyMuPDF, pdfplumber, Tesseract |
12
+ | **Shredding** | Requirement extraction, Section L/M parsing | SpaCy, Regex, Groq (Llama) |
13
+ | **Knowledge** | Vector storage, semantic search, RAG | ChromaDB, OpenAI, Groq (OSS) |
14
+ | **Analysis** | Risk scoring, Go/No-Go decisions | Llama 3.3 70B (Groq) |
15
+ | **Assembly** | Document generation, form filling | docxtpl, openpyxl, PyMuPDF |
16
+ | **Visual** | Gantt charts, timeline visualization | Matplotlib, Plotly |
17
+ | **Tasks** | Sequential AI processing & Orchestration | Celery, Redis, BackgroundTasks |
18
+
19
+ ## AI Provider Strategy
20
+
21
+ The platform utilizes a **Multi-Provider Fallback System** to ensure high availability and performance:
22
+
23
+ 1. **Primary**: **Groq** (Ultra-low latency hardware-accelerated inference using Llama 3.3 70B and Llama 3.1 8B)
24
+ 2. **Secondary**: **OpenRouter** (Unified API for various OSS and Frontier models)
25
+ 3. **Tertiary**: **OpenAI** (Advanced reasoning & primary fallback)
26
+ 4. **Resiliency**: **BackgroundTasks Fallback** (Automatically switches to local FastAPI background tasks if Redis/Celery is unavailable)
27
+
28
+ ## Project Structure
29
+
30
+ ```
31
+ contravaulthvnkz/
32
+ ├── backend/
33
+ │ ├── app/
34
+ │ │ ├── main.py # FastAPI application
35
+ │ │ ├── core/config.py # Configuration settings
36
+ │ │ ├── ingestion/ # Document ingestion
37
+ │ │ ├── shredding/ # Requirement extraction
38
+ │ │ ├── knowledge/ # Vector DB & RAG
39
+ │ │ ├── integration/ # SAM.gov integration
40
+ │ │ └── visual/ # Visualization
41
+ │ ├── requirements.txt
42
+ │ └── .env
43
+ ├── app/ # Next.js App Router
44
+ │ ├── layout.tsx # Root layout & design system
45
+ │ ├── page.tsx # Home/Dashboard
46
+ │ └── [route]/page.tsx # Page wrappers for views
47
+ ├── views/ # Pure React views (Design System)
48
+ │ ├── Dashboard.tsx
49
+ │ ├── Ingestion.tsx
50
+ │ ├── Compliance.tsx
51
+ │ ├── Editor.tsx
52
+ │ ├── ExcelPricing.tsx
53
+ │ ├── Submission.tsx
54
+ │ ├── VectorOps.tsx
55
+ │ └── Settings.tsx
56
+ ├── components/ # Reusable components
57
+ ├── services/api.ts # Frontend API client
58
+ └── package.json
59
+ ```
60
+
61
+ ## Quick Start
62
+
63
+ ### Backend Setup
64
+
65
+ ```bash
66
+ cd backend
67
+
68
+ # Create virtual environment
69
+ python -m venv venv
70
+ source venv/bin/activate # or venv\Scripts\activate on Windows
71
+
72
+ # Install dependencies
73
+ pip install -r requirements.txt
74
+
75
+ # Download spaCy model
76
+ python -m spacy download en_core_web_sm
77
+
78
+ # Run the server
79
+ uvicorn app.main:app --reload --port 8000
80
+ ```
81
+
82
+ ### Frontend Setup
83
+
84
+ ```bash
85
+ # Install dependencies
86
+ npm install
87
+
88
+ # Start development server
89
+ npm run dev
90
+ ```
91
+
92
+ The frontend runs on `http://localhost:3000` and the API on `http://localhost:8000`.
93
+
94
+ ## Deployment
95
+
96
+ ### Backend (Render)
97
+
98
+ The backend is configured for deployment on **Render** using the provided `render.yaml` blueprint.
99
+
100
+ 1. Connect your GitHub repository to Render.
101
+ 2. Render will automatically detect the `render.yaml` file and prompt you to create the "Blueprint".
102
+ 3. Fill in the required environment variables (API Keys, Database URL).
103
+ 4. The service will be deployed as a Python web service under the root directory `backend/`.
104
+
105
+ ### Frontend (Vercel)
106
+
107
+ The frontend is ready for **Vercel** deployment.
108
+
109
+ 1. Push your code to GitHub.
110
+ 2. Import the project in Vercel.
111
+ 3. Add the environment variable `NEXT_PUBLIC_API_URL` pointing to your deployed Render API (e.g., `https://contravaulthvnkz-api.onrender.com/api/v1`).
112
+ 4. Deploy!
113
+
114
+ ## API Endpoints
115
+
116
+ ### Document Processing
117
+
118
+ - `POST /api/v1/ingest` - Ingest PDF/DOCX document
119
+ - `POST /api/v1/shred` - Extract requirements from document
120
+ - `GET /api/v1/requirements/{project_id}` - Get project requirements
121
+
122
+ ### Analysis
123
+
124
+ - `POST /api/v1/analyze/go-no-go` - Perform Go/No-Go analysis
125
+ - `POST /api/v1/analyze/risk/{project_id}` - Analyze project risk
126
+ - `POST /api/v1/analyze/win-themes/{project_id}` - Generate win themes
127
+
128
+ ### SAM.gov Integration
129
+
130
+ - `POST /api/v1/sam/search` - Search federal opportunities
131
+ - `GET /api/v1/sam/opportunity/{notice_id}` - Get full details
132
+ - `POST /api/v1/sam/download/{notice_id}` - Download solicitation artifacts
133
+
134
+ ### Knowledge Engine
135
+
136
+ - `POST /api/v1/knowledge/index` - Index document to vector store
137
+ - `POST /api/v1/knowledge/search` - Semantic search
138
+ - `POST /api/v1/knowledge/ask` - RAG question answering
139
+
140
+ ### Document Generation
141
+
142
+ - `POST /api/v1/generate/proposal` - Generate Word document
143
+ - `POST /api/v1/generate/compliance-matrix/{project_id}` - Generate Excel matrix
144
+
145
+ ### Visualization
146
+
147
+ - `POST /api/v1/visual/gantt` - Generate Gantt chart
148
+ - `GET /api/v1/visual/gantt/{project_id}` - Get project Gantt (PNG)
149
+
150
+ ## Configuration
151
+
152
+ ### Backend (.env)
153
+
154
+ ```
155
+ OPENAI_API_KEY=sk-...
156
+ SAM_GOV_API_KEY=...
157
+ PINECONE_API_KEY=...
158
+ CHROMA_PERSIST_DIR=./chroma_db
159
+ EMBEDDING_MODEL=text-embedding-3-small
160
+ ```
161
+
162
+ ### Frontend (.env.local)
163
+
164
+ ```
165
+ NEXT_PUBLIC_API_URL=http://localhost:8000/api/v1
166
+ ```
167
+
168
+ ## Key Features
169
+
170
+ ### Premium Visual Design System
171
+
172
+ - **Rich Aesthetics**: High-contrast dark mode support, glassmorphism UI elements, and vibrant accent colors.
173
+ - **Fluid Motion**: Logic-heavy views (Editor, Excel, Vectors) feature high-FPS entrance animations and hover transitions.
174
+ - **Dynamic Interaction**: All buttons and interactive cards utilize `active:scale-95` feedback for a tactile feel.
175
+
176
+ ### Hybrid Requirement Extraction
177
+
178
+ 1. **Layer 1 (Regex)**: Pattern matching for modal verbs (shall, must, will)
179
+ 2. **Layer 2 (LLM)**: Semantic filtering for contractor obligations
180
+
181
+ ### Section Recognition
182
+
183
+ - **Section L**: Instructions (format, page limits)
184
+ - **Section M**: Evaluation Factors (scoring)
185
+ - **Section C**: Statement of Work
186
+
187
+ ### Compliance Matrix
188
+
189
+ Generated Excel with requirement ID, risk scoring, volume assignment, and status tracking.
190
+
191
+ ### Gantt Charts
192
+
193
+ Back-casting algorithm from submission date through review milestones.
194
+
195
+ ## Frontend Workspace
196
+
197
+ | View | Route | Function |
198
+ | ------------ | ------------- | -------------------------------- |
199
+ | Dashboard | `/` | Overview & project list |
200
+ | Workbench | `/ingestion` | Document upload & analysis |
201
+ | Compliance | `/compliance` | Requirements matrix |
202
+ | Authoring | `/editor` | Authoring with AI Copilot |
203
+ | Pricing | `/pricing` | Excel-like cost editor |
204
+ | Submission | `/submission` | Final package & sanitization |
205
+ | Vector Index | `/vectors` | Semantic embedding visualization |
206
+ | Settings | `/settings` | Global configuration |
207
+
208
+ ## Interactive API Docs
209
+
210
+ Once the backend is running, visit:
211
+
212
+ - Swagger UI: `http://localhost:8000/docs`
213
+ - ReDoc: `http://localhost:8000/redoc`
app/compliance/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Compliance from '@/views/Compliance';
2
+
3
+ export default function Page() {
4
+ return <Compliance />;
5
+ }
app/deep-research/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import DeepResearch from "@/views/DeepResearch";
2
+
3
+ export default function DeepResearchPage() {
4
+ return <DeepResearch />;
5
+ }
app/editor/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Editor from '@/views/Editor';
2
+
3
+ export default function Page() {
4
+ return <Editor />;
5
+ }
app/globals.css ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @tailwind base;
2
+ @tailwind components;
3
+ @tailwind utilities;
4
+
5
+ /* Material Symbols Icons - Critical styles for proper icon rendering */
6
+ @layer base {
7
+ .material-symbols-outlined {
8
+ font-family: 'Material Symbols Outlined';
9
+ font-weight: normal;
10
+ font-style: normal;
11
+ font-size: 24px;
12
+ line-height: 1;
13
+ letter-spacing: normal;
14
+ text-transform: none;
15
+ display: inline-block;
16
+ white-space: nowrap;
17
+ word-wrap: normal;
18
+ direction: ltr;
19
+ -webkit-font-feature-settings: 'liga';
20
+ font-feature-settings: 'liga';
21
+ -webkit-font-smoothing: antialiased;
22
+ -moz-osx-font-smoothing: grayscale;
23
+ font-variation-settings: 'FILL' 0, 'wght' 400, 'GRAD' 0, 'opsz' 24;
24
+ }
25
+
26
+ /* Filled variant */
27
+ .material-symbols-outlined.filled {
28
+ font-variation-settings: 'FILL' 1, 'wght' 400, 'GRAD' 0, 'opsz' 24;
29
+ }
30
+ }
31
+
32
+ @layer components {
33
+ .btn {
34
+ @apply inline-flex items-center justify-center gap-2 px-4 py-2 rounded-lg text-sm font-medium transition-all duration-200 active:scale-95 disabled:opacity-50 disabled:pointer-events-none;
35
+ }
36
+
37
+ .btn-primary {
38
+ @apply bg-primary text-white hover:bg-primary-dark shadow-md shadow-primary/20;
39
+ }
40
+
41
+ .btn-secondary {
42
+ @apply bg-white dark:bg-slate-800 border border-slate-200 dark:border-slate-700 text-slate-700 dark:text-slate-300 hover:bg-slate-50 dark:hover:bg-slate-700;
43
+ }
44
+
45
+ .card {
46
+ @apply bg-white dark:bg-panel-dark border border-slate-200 dark:border-slate-800 rounded-xl shadow-sm transition-all duration-300 hover:shadow-md;
47
+ }
48
+ }
49
+
50
+ /* Custom Scrollbar */
51
+ ::-webkit-scrollbar {
52
+ width: 8px;
53
+ height: 8px;
54
+ }
55
+ ::-webkit-scrollbar-track {
56
+ background: transparent;
57
+ }
58
+ ::-webkit-scrollbar-thumb {
59
+ background: #cbd5e1;
60
+ border-radius: 4px;
61
+ }
62
+ ::-webkit-scrollbar-thumb:hover {
63
+ background: #94a3b8;
64
+ }
65
+ .dark ::-webkit-scrollbar-thumb {
66
+ background: #475569;
67
+ }
68
+ .dark ::-webkit-scrollbar-thumb:hover {
69
+ background: #64748b;
70
+ }
71
+
72
+ /* Specialized Utilities */
73
+ .glass-panel {
74
+ background: rgba(30, 41, 59, 0.6);
75
+ backdrop-filter: blur(12px);
76
+ border: 1px solid rgba(255, 255, 255, 0.1);
77
+ }
78
+
79
+ .animate-in {
80
+ animation: fadeIn 0.4s ease-out forwards;
81
+ }
82
+
83
+ @keyframes fadeIn {
84
+ from { opacity: 0; transform: translateY(10px); }
85
+ to { opacity: 1; transform: translateY(0); }
86
+ }
app/ingestion/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Ingestion from '@/views/Ingestion';
2
+
3
+ export default function Page() {
4
+ return <Ingestion />;
5
+ }
app/layout.tsx ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import React from 'react';
2
+ import type { Metadata } from 'next';
3
+ import { Inter, Merriweather, Roboto_Mono } from 'next/font/google';
4
+ import './globals.css';
5
+ import { Sidebar } from '@/components/Sidebar';
6
+
7
+ const inter = Inter({ subsets: ['latin'], variable: '--font-inter', display: 'swap' });
8
+ const merriweather = Merriweather({ weight: ['300', '400', '700'], subsets: ['latin'], variable: '--font-merriweather', display: 'swap' });
9
+ const robotoMono = Roboto_Mono({ subsets: ['latin'], variable: '--font-roboto-mono', display: 'swap' });
10
+
11
+ export const metadata: Metadata = {
12
+ title: 'RFP Auto - Enterprise Proposal Platform',
13
+ description: 'Enterprise Proposal Platform',
14
+ };
15
+
16
+ export default function RootLayout({
17
+ children,
18
+ }: {
19
+ children: React.ReactNode;
20
+ }) {
21
+ return (
22
+ <html lang="en" suppressHydrationWarning className={`${inter.variable} ${merriweather.variable} ${robotoMono.variable} h-full`}>
23
+ <head>
24
+ <link rel="preconnect" href="https://fonts.googleapis.com" />
25
+ <link rel="preconnect" href="https://fonts.gstatic.com" crossOrigin="anonymous" />
26
+ <link
27
+ href="https://fonts.googleapis.com/css2?family=Material+Symbols+Outlined:opsz,wght,FILL,GRAD@20..48,100..700,0..1,-50..200&display=swap"
28
+ rel="stylesheet"
29
+ />
30
+ </head>
31
+ <body className="bg-background-light dark:bg-background-dark text-slate-900 dark:text-white font-sans antialiased overflow-hidden selection:bg-primary/20 h-full">
32
+ <div className="flex h-screen w-full overflow-hidden">
33
+ <Sidebar />
34
+ <main className="flex-1 overflow-hidden relative">
35
+ {children}
36
+ </main>
37
+ </div>
38
+ </body>
39
+ </html>
40
+ );
41
+ }
app/opportunities/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Opportunities from '@/views/Opportunities';
2
+
3
+ export default function Page() {
4
+ return <Opportunities />;
5
+ }
app/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Dashboard from '@/views/Dashboard';
2
+
3
+ export default function Page() {
4
+ return <Dashboard />;
5
+ }
app/pricing/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import ExcelPricing from '@/views/ExcelPricing';
2
+
3
+ export default function Page() {
4
+ return <ExcelPricing />;
5
+ }
app/projects/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Projects from "@/views/Projects";
2
+
3
+ export default function ProjectsPage() {
4
+ return <Projects />;
5
+ }
app/settings/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Settings from '@/views/Settings';
2
+
3
+ export default function Page() {
4
+ return <Settings />;
5
+ }
app/submission/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import Submission from '@/views/Submission';
2
+
3
+ export default function Page() {
4
+ return <Submission />;
5
+ }
app/vectors/page.tsx ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ import VectorOps from '@/views/VectorOps';
2
+
3
+ export default function Page() {
4
+ return <VectorOps />;
5
+ }
backend/.env.example ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # ContraVault Environment Configuration
2
+ # Copy this file to .env and fill in your values
3
+
4
+ # =============================================================================
5
+ # AI PROVIDERS (All Free Tiers)
6
+ # =============================================================================
7
+
8
+ # OpenRouter - Primary provider for most features
9
+ # Get free API key at: https://openrouter.ai/keys
10
+ OPENROUTER_API_KEY=sk-or-v1-your-key-here
11
+
12
+ # Cerebras - Fast inference for knowledge QA
13
+ # Get free API key at: https://cloud.cerebras.ai/
14
+ CEREBRAS_API_KEY=csk-your-key-here
15
+
16
+ # Mistral - Code generation (optional)
17
+ # Get free API key at: https://console.mistral.ai/
18
+ MISTRAL_API_KEY=your-key-here
19
+
20
+ # Groq - Alternative fast inference (legacy)
21
+ # Get free API key at: https://console.groq.com/
22
+ GROQ_API_KEY=gsk_your-key-here
23
+
24
+ # =============================================================================
25
+ # DATABASE
26
+ # =============================================================================
27
+
28
+ # PostgreSQL connection (Neon recommended)
29
+ DATABASE_URL=postgresql://user:password@host/database?sslmode=require
30
+ ASYNC_DATABASE_URL=postgresql+asyncpg://user:password@host/database?ssl=require
31
+
32
+ # =============================================================================
33
+ # VECTOR DATABASE
34
+ # =============================================================================
35
+
36
+ # Local ChromaDB storage path
37
+ CHROMA_PERSIST_DIR=./chroma_db
38
+
39
+ # =============================================================================
40
+ # EMBEDDINGS
41
+ # =============================================================================
42
+
43
+ # OpenAI embeddings (uses OpenRouter by default)
44
+ OPENAI_API_KEY=sk-or-v1-your-openrouter-key
45
+ OPENAI_BASE_URL=https://openrouter.ai/api/v1
46
+ EMBEDDING_MODEL=text-embedding-3-small
47
+ EMBEDDING_DIMENSIONS=1536
48
+
49
+ # =============================================================================
50
+ # STORAGE
51
+ # =============================================================================
52
+
53
+ UPLOAD_DIR=./uploads
54
+ OUTPUT_DIR=./outputs
55
+ TEMPLATE_DIR=./templates
56
+
57
+ # =============================================================================
58
+
59
+ # =============================================================================
60
+ # NLI / ML SETTINGS
61
+ # =============================================================================
62
+
63
+ USE_TRANSFORMER_NLI=false
64
+ CROSS_ENCODER_MODEL=balanced
65
+
66
+ # =============================================================================
67
+ # EXTERNAL SERVICES (Optional)
68
+ # =============================================================================
69
+
70
+ FIRECRAWL_API_KEY=your-key-here
backend/README.md ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Next-Gen RFP Automation Backend
2
+
3
+ ## Overview
4
+
5
+ This backend implements the "Executive Architecture" for the RFP Automation platform. It is designed as a modular Python system with specialized engines for ingestion, analysis, and generation.
6
+
7
+ ## Setup Instructions
8
+
9
+ ### Prerequisites
10
+
11
+ - Python 3.10+
12
+ - Tesseract OCR (installed and in PATH)
13
+ - C++ Build Tools (for some NLP libraries)
14
+
15
+ ### Installation
16
+
17
+ 1. Create a virtual environment:
18
+
19
+ ```powershell
20
+ python -m venv venv
21
+ .\venv\Scripts\Activate
22
+ ```
23
+
24
+ 2. Install dependencies:
25
+
26
+ ```powershell
27
+ pip install -r requirements.txt
28
+ ```
29
+
30
+ _Note: `detectron2` and `unstructured` are heavy dependencies. If installation fails, try installing strictly the core packages first or use pre-built wheels._
31
+
32
+ 3. Configure Environment:
33
+ The `.env` file has been created with your provided API keys. Ensure it remains secure.
34
+
35
+ ### Running the Server
36
+
37
+ ```powershell
38
+ uvicorn app.main:app --reload
39
+ ```
40
+
41
+ ## Module Structure
42
+
43
+ 1. **Ingestion Engine** (`app/ingestion`):
44
+ - Handles PDF/Word parsing.
45
+ - Implements partitioning strategies (Hi-Res, Fast, OCR).
46
+
47
+ 2. **Shredding Engine** (`app/shredding`):
48
+ - Hybrid extraction model (Regex + Agentic).
49
+ - Generates `ComplianceItem` objects.
50
+
51
+ 3. **Knowledge Engine** (`app/knowledge`):
52
+ - Manages Vector DB connections (ChromaDB/pgvector).
53
+ - Handles semantic chunking and retrieval.
54
+
55
+ 4. **Analysis Engine** (`app/analysis`):
56
+ - Calculates Risk Scores ("Poison Pills").
57
+ - Detects contradictions.
58
+
59
+ 5. **Assembly Engine** (`app/assembly`):
60
+ - Generates Excel Compliance Matrices.
61
+ - Generates Word Proposals and fills PDF forms.
62
+
63
+ 6. **Visual Engine** (`app/visual`):
64
+ - Generates Gantt charts for proposal management.
65
+
66
+ ## API Documentation
67
+
68
+ Once running, visit `http://localhost:8000/docs` for the interactive API documentation.
backend/ai_router.py ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import requests
3
+ import json
4
+ import logging
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+ class AIModelRouter:
9
+ def __init__(self):
10
+ self.openrouter_key = os.getenv("OPENROUTER_API_KEY")
11
+ self.cerebras_key = os.getenv("CEREBRAS_API_KEY")
12
+
13
+ # Priority List: High Intelligence -> High Speed -> Fallback
14
+ self.model_priority = [
15
+ # PRIMARY: Google Gemma 3 27B (High Context, Multimodal, Free on OpenRouter)
16
+ {"id": "google/gemma-3-27b-instruct:free", "provider": "openrouter", "context": 131000},
17
+
18
+ # SECONDARY: Llama 3.3 70B (High Intelligence, Free on OpenRouter)
19
+ {"id": "meta-llama/llama-3.3-70b-instruct:free", "provider": "openrouter", "context": 64000},
20
+
21
+ # TERTIARY: Llama 3.1 8B (Super Fast, Free on Cerebras)
22
+ {"id": "llama3.1-8b", "provider": "cerebras", "context": 8192},
23
+
24
+ # FALLBACK: Mistral Small (Generalist)
25
+ {"id": "mistralai/mistral-small-24b-instruct-2501:free", "provider": "openrouter", "context": 32000}
26
+ ]
27
+
28
+ def generate_response(self, system_prompt, user_content):
29
+ errors = []
30
+
31
+ for model in self.model_priority:
32
+ try:
33
+ print(f"🔄 Routing to Model: {model['id']} via {model['provider']}...")
34
+
35
+ if model['provider'] == "openrouter":
36
+ response = self._call_openrouter(model['id'], system_prompt, user_content)
37
+ elif model['provider'] == "cerebras":
38
+ response = self._call_cerebras(model['id'], system_prompt, user_content)
39
+
40
+ if response:
41
+ print(f"✅ Success with {model['id']}")
42
+ return {
43
+ "content": response,
44
+ "model_used": model['id'],
45
+ "provider": model['provider']
46
+ }
47
+ except Exception as e:
48
+ print(f"❌ Failed {model['id']}: {str(e)}")
49
+ errors.append(f"{model['id']}: {str(e)}")
50
+ continue # Auto-swap to next model
51
+
52
+ raise Exception(f"All AI Models Failed. Errors: {errors}")
53
+
54
+ def _call_openrouter(self, model_id, system, user):
55
+ headers = {
56
+ "Authorization": f"Bearer {self.openrouter_key}",
57
+ "HTTP-Referer": "https://heavenkeys.ca",
58
+ "X-Title": "ContractVault"
59
+ }
60
+ payload = {
61
+ "model": model_id,
62
+ "messages": [
63
+ {"role": "system", "content": system},
64
+ {"role": "user", "content": user}
65
+ ]
66
+ }
67
+ resp = requests.post("https://openrouter.ai/api/v1/chat/completions", json=payload, headers=headers, timeout=120)
68
+ resp.raise_for_status()
69
+ return resp.json()['choices'][0]['message']['content']
70
+
71
+ def _call_cerebras(self, model_id, system, user):
72
+ headers = {"Authorization": f"Bearer {self.cerebras_key}"}
73
+ payload = {
74
+ "model": model_id,
75
+ "stream": False,
76
+ "messages": [
77
+ {"role": "system", "content": system},
78
+ {"role": "user", "content": user}
79
+ ]
80
+ }
81
+ resp = requests.post("https://api.cerebras.ai/v1/chat/completions", json=payload, headers=headers, timeout=120)
82
+ resp.raise_for_status()
83
+ return resp.json()['choices'][0]['message']['content']
backend/app/__init__.py ADDED
@@ -0,0 +1 @@
 
 
1
+ # RFP Automation Platform - Main Package
backend/app/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (149 Bytes). View file
 
backend/app/__pycache__/database.cpython-313.pyc ADDED
Binary file (1.54 kB). View file
 
backend/app/__pycache__/main.cpython-313.pyc ADDED
Binary file (88 kB). View file
 
backend/app/__pycache__/models.cpython-313.pyc ADDED
Binary file (60.6 kB). View file
 
backend/app/__pycache__/tasks.cpython-313.pyc ADDED
Binary file (8.66 kB). View file
 
backend/app/ai/__init__.py ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # AI Module - Multi-Provider with Smart Routing
2
+ # Supports OpenRouter (free), Cerebras (free), Mistral (free)
3
+
4
+ # Import from smart_router (new implementation with auto-swap)
5
+ from .smart_router import (
6
+ SmartAIRouter,
7
+ AIProvider,
8
+ FeatureType,
9
+ ChatMessage,
10
+ AIResponse,
11
+ ModelConfig,
12
+ ModelChain,
13
+ ProviderHealth,
14
+ SwapReason,
15
+ ALL_MODELS,
16
+ FEATURE_MODEL_CHAINS,
17
+ get_smart_router,
18
+ configure_smart_router,
19
+ get_model_by_id,
20
+ # Backwards compatibility aliases
21
+ MultiProviderAI,
22
+ get_ai_service,
23
+ configure_ai_service,
24
+ )
25
+
26
+ # Also import legacy for explicit access
27
+ from . import providers as legacy_providers
28
+
29
+ # Legacy alias for AVAILABLE_MODELS
30
+ AVAILABLE_MODELS = ALL_MODELS
31
+
32
+ __all__ = [
33
+ # New smart router
34
+ "SmartAIRouter",
35
+ "AIProvider",
36
+ "FeatureType",
37
+ "ChatMessage",
38
+ "AIResponse",
39
+ "ModelConfig",
40
+ "ModelChain",
41
+ "ProviderHealth",
42
+ "SwapReason",
43
+ "ALL_MODELS",
44
+ "AVAILABLE_MODELS",
45
+ "FEATURE_MODEL_CHAINS",
46
+ "get_smart_router",
47
+ "configure_smart_router",
48
+ "get_model_by_id",
49
+ # Backwards compatibility
50
+ "MultiProviderAI",
51
+ "get_ai_service",
52
+ "configure_ai_service",
53
+ "legacy_providers",
54
+ ]
backend/app/ai/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (828 Bytes). View file
 
backend/app/ai/__pycache__/providers.cpython-313.pyc ADDED
Binary file (25.3 kB). View file
 
backend/app/ai/__pycache__/smart_router.cpython-313.pyc ADDED
Binary file (42.8 kB). View file
 
backend/app/ai/providers.py ADDED
@@ -0,0 +1,533 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Multi-Provider AI Service
2
+ # Supports Groq, OpenRouter, and OpenAI with automatic fallback
3
+
4
+ import os
5
+ import json
6
+ import httpx
7
+ import asyncio
8
+ from typing import Optional, List, Dict, Any, AsyncGenerator
9
+ from pydantic import BaseModel
10
+ from enum import Enum
11
+ import logging
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ class AIProvider(str, Enum):
17
+ GROQ = "groq"
18
+ OPENROUTER = "openrouter"
19
+ OPENAI = "openai"
20
+ CEREBRAS = "cerebras"
21
+ MISTRAL = "mistral"
22
+
23
+
24
+ class ModelConfig(BaseModel):
25
+ provider: AIProvider
26
+ model_id: str
27
+ display_name: str
28
+ context_window: int
29
+ supports_streaming: bool = True
30
+ supports_vision: bool = False
31
+ is_free: bool = False
32
+
33
+
34
+ # Available models per provider
35
+ AVAILABLE_MODELS: Dict[AIProvider, List[ModelConfig]] = {
36
+ AIProvider.CEREBRAS: [
37
+ ModelConfig(provider=AIProvider.CEREBRAS, model_id="llama-3.3-70b", display_name="Cerebras Llama 3.3 70B", context_window=65536),
38
+ ModelConfig(provider=AIProvider.CEREBRAS, model_id="llama3.1-8b", display_name="Cerebras Llama 3.1 8B", context_window=8192),
39
+ ModelConfig(provider=AIProvider.CEREBRAS, model_id="qwen-3-32b", display_name="Cerebras Qwen 3 32B", context_window=65536),
40
+ ],
41
+ AIProvider.GROQ: [
42
+ ModelConfig(provider=AIProvider.GROQ, model_id="llama-3.3-70b-versatile", display_name="Groq Llama 3.3 70B", context_window=128000),
43
+ ModelConfig(provider=AIProvider.GROQ, model_id="llama-3.1-8b-instant", display_name="Groq Llama 3.1 8B", context_window=128000),
44
+ ],
45
+ AIProvider.OPENROUTER: [
46
+ ModelConfig(provider=AIProvider.OPENROUTER, model_id="meta-llama/llama-3.3-70b-instruct:free", display_name="Llama 3.3 70B (Free)", context_window=131072, is_free=True),
47
+ ModelConfig(provider=AIProvider.OPENROUTER, model_id="google/gemma-3-27b-instruct:free", display_name="Gemma 3 27B (Free)", context_window=131072, is_free=True),
48
+ ModelConfig(provider=AIProvider.OPENROUTER, model_id="nousresearch/hermes-3-405b:free", display_name="Hermes 3 405B (Free)", context_window=131072, is_free=True),
49
+ ModelConfig(provider=AIProvider.OPENROUTER, model_id="google/gemma-3-12b-instruct:free", display_name="Gemma 3 12B (Free)", context_window=32768, is_free=True),
50
+ ModelConfig(provider=AIProvider.OPENROUTER, model_id="meta-llama/llama-3.2-3b-instruct:free", display_name="Llama 3.2 3B (Free)", context_window=131072, is_free=True),
51
+ ModelConfig(provider=AIProvider.OPENROUTER, model_id="qwen/qwen2.5-vl-7b-instruct:free", display_name="Qwen 2.5 VL 7B (Free)", context_window=32768, is_free=True),
52
+ ],
53
+ AIProvider.MISTRAL: [
54
+ ModelConfig(provider=AIProvider.MISTRAL, model_id="mistral-large-latest", display_name="Mistral Large 3", context_window=128000),
55
+ ModelConfig(provider=AIProvider.MISTRAL, model_id="mistral-small-latest", display_name="Mistral Small 3.2", context_window=128000),
56
+ ModelConfig(provider=AIProvider.MISTRAL, model_id="codestral-latest", display_name="Codestral", context_window=32768),
57
+ ],
58
+ AIProvider.OPENAI: [
59
+ ModelConfig(provider=AIProvider.OPENAI, model_id="gpt-4o", display_name="GPT-4o", context_window=128000),
60
+ ModelConfig(provider=AIProvider.OPENAI, model_id="gpt-4o-mini", display_name="GPT-4o Mini", context_window=128000),
61
+ ],
62
+ }
63
+
64
+
65
+ class ChatMessage(BaseModel):
66
+ role: str # "system", "user", "assistant"
67
+ content: str
68
+
69
+
70
+ class AIResponse(BaseModel):
71
+ content: str
72
+ model: str
73
+ provider: AIProvider
74
+ usage: Optional[Dict[str, int]] = None
75
+ finish_reason: Optional[str] = None
76
+
77
+
78
+ class MultiProviderAI:
79
+ """
80
+ Multi-provider AI service supporting Cerebras, Groq, OpenRouter, Mistral, and OpenAI.
81
+ Provides automatic fallback between providers and specific routing for RFP features.
82
+ """
83
+
84
+ def __init__(
85
+ self,
86
+ groq_api_key: Optional[str] = None,
87
+ openrouter_api_key: Optional[str] = None,
88
+ openai_api_key: Optional[str] = None,
89
+ cerebras_api_key: Optional[str] = None,
90
+ mistral_api_key: Optional[str] = None,
91
+ default_provider: AIProvider = AIProvider.CEREBRAS,
92
+ default_model: Optional[str] = None,
93
+ ):
94
+ from app.core.config import settings
95
+
96
+ self.api_keys = {
97
+ AIProvider.CEREBRAS: cerebras_api_key or settings.CEREBRAS_API_KEY or os.getenv("CEREBRAS_API_KEY", ""),
98
+ AIProvider.GROQ: groq_api_key or settings.GROQ_API_KEY or os.getenv("GROQ_API_KEY", ""),
99
+ AIProvider.OPENROUTER: openrouter_api_key or settings.OPENROUTER_API_KEY or os.getenv("OPENROUTER_API_KEY", ""),
100
+ AIProvider.MISTRAL: mistral_api_key or settings.MISTRAL_API_KEY or os.getenv("MISTRAL_API_KEY", ""),
101
+ AIProvider.OPENAI: openai_api_key or settings.OPENAI_API_KEY or os.getenv("OPENAI_API_KEY", ""),
102
+ }
103
+
104
+ self.base_urls = {
105
+ AIProvider.CEREBRAS: "https://api.cerebras.ai/v1",
106
+ AIProvider.GROQ: "https://api.groq.com/openai/v1",
107
+ AIProvider.OPENROUTER: "https://openrouter.ai/api/v1",
108
+ AIProvider.MISTRAL: "https://api.mistral.ai/v1",
109
+ AIProvider.OPENAI: "https://api.openai.com/v1",
110
+ }
111
+
112
+ # Select a configured provider if default is not configured
113
+ if not self.api_keys.get(default_provider):
114
+ for p in [AIProvider.CEREBRAS, AIProvider.GROQ, AIProvider.OPENROUTER, AIProvider.MISTRAL, AIProvider.OPENAI]:
115
+ if self.api_keys.get(p):
116
+ default_provider = p
117
+ break
118
+
119
+ self.default_provider = default_provider
120
+ self.default_model = default_model or self._get_default_model(default_provider)
121
+
122
+ # Provider priority for general fallback
123
+ self.fallback_order = [AIProvider.CEREBRAS, AIProvider.GROQ, AIProvider.OPENROUTER, AIProvider.MISTRAL, AIProvider.OPENAI]
124
+
125
+ # Feature-to-Provider mapping (Primary and Fallback)
126
+ self.feature_routing = {
127
+ "shredding": [(AIProvider.CEREBRAS, "llama-3.3-70b"), (AIProvider.OPENROUTER, "meta-llama/llama-3.3-70b-instruct:free")],
128
+ "analysis": [(AIProvider.CEREBRAS, "llama-3.3-70b"), (AIProvider.OPENROUTER, "meta-llama/llama-3.3-70b-instruct:free")],
129
+ "blind_rfp": [(AIProvider.OPENROUTER, "google/gemma-3-27b-instruct:free"), (AIProvider.CEREBRAS, "qwen-3-32b")],
130
+ "contradictions": [(AIProvider.OPENROUTER, "google/gemma-3-27b-instruct:free"), (AIProvider.MISTRAL, "mistral-small-latest")],
131
+ "win_themes": [(AIProvider.OPENROUTER, "nousresearch/hermes-3-405b:free"), (AIProvider.MISTRAL, "mistral-large-latest")],
132
+ "rfi_gen": [(AIProvider.CEREBRAS, "llama-3.3-70b"), (AIProvider.OPENROUTER, "meta-llama/llama-3.3-70b-instruct:free")],
133
+ "chat": [(AIProvider.CEREBRAS, "llama3.1-8b"), (AIProvider.OPENROUTER, "meta-llama/llama-3.2-3b-instruct:free")],
134
+ }
135
+
136
+ def _get_default_model(self, provider: AIProvider) -> str:
137
+ models = AVAILABLE_MODELS.get(provider, [])
138
+ return models[0].model_id if models else ""
139
+
140
+ def _get_headers(self, provider: AIProvider) -> Dict[str, str]:
141
+ headers = {
142
+ "Content-Type": "application/json",
143
+ "Authorization": f"Bearer {self.api_keys[provider]}",
144
+ }
145
+
146
+ if provider == AIProvider.OPENROUTER:
147
+ headers["HTTP-Referer"] = "https://contravault.ai"
148
+ headers["X-Title"] = "ContraVault RFP Platform"
149
+
150
+ return headers
151
+
152
+ def get_available_providers(self) -> List[Dict[str, Any]]:
153
+ """Get list of available providers with their API key status."""
154
+ providers = []
155
+ for provider in AIProvider:
156
+ has_key = bool(self.api_keys.get(provider))
157
+ models = AVAILABLE_MODELS.get(provider, [])
158
+ providers.append({
159
+ "id": provider.value,
160
+ "name": provider.value.title(),
161
+ "configured": has_key,
162
+ "models": [m.model_dump() for m in models],
163
+ })
164
+ return providers
165
+
166
+ def get_available_models(self, provider: Optional[AIProvider] = None) -> List[ModelConfig]:
167
+ """Get available models for a specific provider or all providers."""
168
+ if provider:
169
+ return AVAILABLE_MODELS.get(provider, [])
170
+
171
+ all_models = []
172
+ for p, models in AVAILABLE_MODELS.items():
173
+ if self.api_keys.get(p):
174
+ all_models.extend(models)
175
+ return all_models
176
+
177
+ async def chat(
178
+ self,
179
+ messages: List[ChatMessage],
180
+ feature: Optional[str] = None,
181
+ provider: Optional[AIProvider] = None,
182
+ model: Optional[str] = None,
183
+ temperature: float = 0.7,
184
+ max_tokens: int = 4096,
185
+ stream: bool = False,
186
+ ) -> AIResponse:
187
+ """
188
+ Send a chat completion request with automatic failover and feature-based routing.
189
+ """
190
+ # Determine initial attempt list
191
+ attempts = []
192
+
193
+ if feature and feature in self.feature_routing:
194
+ # Add feature-specific route first
195
+ for p, m in self.feature_routing[feature]:
196
+ if self.api_keys.get(p):
197
+ attempts.append((p, m))
198
+
199
+ if provider and model:
200
+ # Use explicit provider/model if requested (but will still failover if fails)
201
+ if (provider, model) not in attempts:
202
+ attempts.append((provider, model))
203
+ elif provider:
204
+ # Use explicit provider with its default model
205
+ m = self._get_default_model(provider)
206
+ if (provider, m) not in attempts:
207
+ attempts.append((provider, m))
208
+
209
+ # Add general fallbacks
210
+ for p in self.fallback_order:
211
+ m = self._get_default_model(p)
212
+ if (p, m) not in attempts:
213
+ attempts.append((p, m))
214
+
215
+ last_error = None
216
+ for p, m in attempts:
217
+ if not self.api_keys.get(p):
218
+ continue
219
+
220
+ try:
221
+ logger.info(f"Attempting chat with {p.value} using model {m} (Feature: {feature or 'None'})")
222
+ if stream:
223
+ return await self._stream_chat(messages, p, m, temperature, max_tokens)
224
+ else:
225
+ return await self._complete_chat(messages, p, m, temperature, max_tokens)
226
+ except Exception as e:
227
+ logger.warning(f"Provider {p.value} ({m}) failed for feature '{feature}': {e}")
228
+ last_error = e
229
+ continue
230
+
231
+ raise Exception(f"All providers failed for {feature or 'chat'}. Last error: {last_error}")
232
+
233
+ async def _complete_chat(
234
+ self,
235
+ messages: List[ChatMessage],
236
+ provider: AIProvider,
237
+ model: str,
238
+ temperature: float,
239
+ max_tokens: int,
240
+ ) -> AIResponse:
241
+ """Make a non-streaming chat completion request."""
242
+ url = f"{self.base_urls[provider]}/chat/completions"
243
+ headers = self._get_headers(provider)
244
+
245
+ payload = {
246
+ "model": model,
247
+ "messages": [{"role": m.role, "content": m.content} for m in messages],
248
+ "temperature": temperature,
249
+ "max_tokens": max_tokens,
250
+ }
251
+
252
+ async with httpx.AsyncClient(timeout=120.0) as client:
253
+ response = await client.post(url, headers=headers, json=payload)
254
+ response.raise_for_status()
255
+ data = response.json()
256
+
257
+ choice = data["choices"][0]
258
+ return AIResponse(
259
+ content=choice["message"]["content"],
260
+ model=data.get("model", model),
261
+ provider=provider,
262
+ usage=data.get("usage"),
263
+ finish_reason=choice.get("finish_reason"),
264
+ )
265
+
266
+ async def _stream_chat(
267
+ self,
268
+ messages: List[ChatMessage],
269
+ provider: AIProvider,
270
+ model: str,
271
+ temperature: float,
272
+ max_tokens: int,
273
+ ) -> AsyncGenerator[str, None]:
274
+ """Make a streaming chat completion request."""
275
+ url = f"{self.base_urls[provider]}/chat/completions"
276
+ headers = self._get_headers(provider)
277
+
278
+ payload = {
279
+ "model": model,
280
+ "messages": [{"role": m.role, "content": m.content} for m in messages],
281
+ "temperature": temperature,
282
+ "max_tokens": max_tokens,
283
+ "stream": True,
284
+ }
285
+
286
+ async with httpx.AsyncClient(timeout=120.0) as client:
287
+ async with client.stream("POST", url, headers=headers, json=payload) as response:
288
+ response.raise_for_status()
289
+ async for line in response.aiter_lines():
290
+ if line.startswith("data: "):
291
+ data_str = line[6:]
292
+ if data_str == "[DONE]":
293
+ break
294
+ try:
295
+ data = json.loads(data_str)
296
+ delta = data["choices"][0].get("delta", {})
297
+ if "content" in delta:
298
+ yield delta["content"]
299
+ except json.JSONDecodeError:
300
+ continue
301
+
302
+ async def analyze_rfp_text(
303
+ self,
304
+ text: str,
305
+ analysis_type: str = "requirements",
306
+ provider: Optional[AIProvider] = None,
307
+ model: Optional[str] = None,
308
+ ) -> Dict[str, Any]:
309
+ """
310
+ Specialized RFP analysis using AI.
311
+ Supports: requirements, risks, go_no_go, win_themes, compliance
312
+ """
313
+ prompts = {
314
+ "requirements": """Analyze the following RFP text and extract all requirements.
315
+ For each requirement, identify:
316
+ 1. The exact text of the requirement
317
+ 2. Whether it's mandatory (shall/must) or desirable (should/may)
318
+ 3. The category (Technical, Management, Legal, Financial, etc.)
319
+ 4. Risk level (High/Medium/Low)
320
+ 5. Any keywords or compliance standards mentioned
321
+
322
+ Return as JSON array with fields: text, type, category, risk_level, keywords
323
+
324
+ RFP Text:
325
+ {text}""",
326
+
327
+ "risks": """Analyze the following RFP text for potential risks and red flags.
328
+ Look for:
329
+ 1. Unlimited liability clauses
330
+ 2. Unrealistic timelines
331
+ 3. Ambiguous scope
332
+ 4. Onerous payment terms
333
+ 5. IP ownership issues
334
+ 6. Performance bond requirements
335
+ 7. Liquidated damages
336
+ 8. Insurance requirements
337
+
338
+ Return as JSON array with fields: risk_type, severity (high/medium/low), clause_text, recommendation
339
+
340
+ RFP Text:
341
+ {text}""",
342
+
343
+ "go_no_go": """Perform a Go/No-Go analysis on this RFP.
344
+ Evaluate:
345
+ 1. Technical fit (1-10)
346
+ 2. Resource availability (1-10)
347
+ 3. Timeline feasibility (1-10)
348
+ 4. Competitive position (1-10)
349
+ 5. Profitability potential (1-10)
350
+ 6. Strategic alignment (1-10)
351
+ 7. Risk level (1-10, lower is better)
352
+
353
+ Provide:
354
+ - Overall recommendation: GO, NO-GO, or CONDITIONAL
355
+ - Confidence percentage
356
+ - Key reasons for the decision
357
+ - Mitigation strategies if CONDITIONAL
358
+
359
+ Return as JSON with fields: recommendation, confidence, scores (object), reasons (array), mitigations (array)
360
+
361
+ RFP Text:
362
+ {text}""",
363
+
364
+ "win_themes": """Based on this RFP, identify potential win themes and discriminators.
365
+ Consider:
366
+ 1. What does the client really need?
367
+ 2. What pain points are implied?
368
+ 3. What differentiators could be emphasized?
369
+ 4. What proof points would be compelling?
370
+
371
+ Return as JSON array with fields: theme_title, description, supporting_evidence, discriminator_level (strong/moderate/weak)
372
+
373
+ RFP Text:
374
+ {text}""",
375
+
376
+ "compliance": """Analyze this RFP for compliance requirements.
377
+ Identify:
378
+ 1. Certifications required (ISO, SOC2, FedRAMP, etc.)
379
+ 2. Security standards
380
+ 3. Regulatory compliance (GDPR, HIPAA, etc.)
381
+ 4. Format requirements (page limits, fonts, sections)
382
+ 5. Submission requirements (deadlines, methods)
383
+
384
+ Return as JSON with fields: certifications (array), security_standards (array), regulations (array), format_requirements (object), submission_requirements (object)
385
+
386
+ RFP Text:
387
+ {text}""",
388
+ }
389
+
390
+ prompt = prompts.get(analysis_type, prompts["requirements"])
391
+ formatted_prompt = prompt.format(text=text[:15000]) # Limit text length
392
+
393
+ messages = [
394
+ ChatMessage(role="system", content="You are an expert RFP analyst. Always respond with valid JSON only, no markdown or explanations."),
395
+ ChatMessage(role="user", content=formatted_prompt),
396
+ ]
397
+
398
+ # Map analysis type to feature for routing
399
+ feature_map = {
400
+ "requirements": "shredding",
401
+ "risks": "analysis",
402
+ "go_no_go": "analysis",
403
+ "win_themes": "win_themes",
404
+ "compliance": "analysis"
405
+ }
406
+ feature = feature_map.get(analysis_type, "analysis")
407
+
408
+ response = await self.chat(messages, feature=feature, provider=provider, model=model, temperature=0.3)
409
+
410
+ # Parse JSON response
411
+ try:
412
+ # Try to extract JSON from response
413
+ content = response.content.strip()
414
+ if content.startswith("```"):
415
+ content = content.split("```")[1]
416
+ if content.startswith("json"):
417
+ content = content[4:]
418
+ return json.loads(content)
419
+ except json.JSONDecodeError:
420
+ return {"raw_response": response.content, "error": "Failed to parse JSON response"}
421
+
422
+ async def generate_proposal_section(
423
+ self,
424
+ section_name: str,
425
+ requirements: List[str],
426
+ company_info: Dict[str, Any],
427
+ style_guide: Optional[str] = None,
428
+ provider: Optional[AIProvider] = None,
429
+ model: Optional[str] = None,
430
+ ) -> str:
431
+ """Generate a proposal section based on requirements and company info."""
432
+ system_prompt = f"""You are an expert proposal writer for government and enterprise RFPs.
433
+ Write professional, compliant proposal content that:
434
+ 1. Directly addresses each requirement
435
+ 2. Uses clear, concise language
436
+ 3. Highlights the company's strengths and experience
437
+ 4. Follows any provided style guidelines
438
+ 5. Maintains a confident but not arrogant tone
439
+
440
+ Company Information:
441
+ {json.dumps(company_info, indent=2)}
442
+
443
+ {"Style Guide: " + style_guide if style_guide else ""}
444
+ """
445
+
446
+ user_prompt = f"""Write the "{section_name}" section for a proposal.
447
+
448
+ Requirements to address:
449
+ {chr(10).join(f"- {r}" for r in requirements)}
450
+
451
+ Generate professional proposal content that addresses each requirement. Include specific details and evidence where appropriate.
452
+ """
453
+
454
+ messages = [
455
+ ChatMessage(role="system", content=system_prompt),
456
+ ChatMessage(role="user", content=user_prompt),
457
+ ]
458
+
459
+ response = await self.chat(messages, provider=provider, model=model, temperature=0.7, max_tokens=4096)
460
+ return response.content
461
+
462
+ async def suggest_improvements(
463
+ self,
464
+ text: str,
465
+ context: str = "proposal",
466
+ provider: Optional[AIProvider] = None,
467
+ model: Optional[str] = None,
468
+ ) -> List[Dict[str, str]]:
469
+ """Suggest improvements for proposal text."""
470
+ prompt = f"""Review the following {context} text and suggest improvements.
471
+ For each suggestion, provide:
472
+ 1. The original text segment
473
+ 2. The suggested improvement
474
+ 3. The reason for the change
475
+ 4. Priority (high/medium/low)
476
+
477
+ Focus on:
478
+ - Clarity and conciseness
479
+ - Compliance with typical RFP requirements
480
+ - Professional tone
481
+ - Specificity and evidence
482
+
483
+ Text to review:
484
+ {text}
485
+
486
+ Return as JSON array with fields: original, suggestion, reason, priority
487
+ """
488
+
489
+ messages = [
490
+ ChatMessage(role="system", content="You are an expert proposal editor. Always respond with valid JSON only."),
491
+ ChatMessage(role="user", content=prompt),
492
+ ]
493
+
494
+ response = await self.chat(messages, provider=provider, model=model, temperature=0.5)
495
+
496
+ try:
497
+ content = response.content.strip()
498
+ if content.startswith("```"):
499
+ content = content.split("```")[1]
500
+ if content.startswith("json"):
501
+ content = content[4:]
502
+ return json.loads(content)
503
+ except json.JSONDecodeError:
504
+ return [{"error": "Failed to parse suggestions", "raw": response.content}]
505
+
506
+
507
+ # Singleton instance
508
+ _ai_service: Optional[MultiProviderAI] = None
509
+
510
+
511
+ def get_ai_service() -> MultiProviderAI:
512
+ """Get or create the AI service singleton."""
513
+ global _ai_service
514
+ if _ai_service is None:
515
+ _ai_service = MultiProviderAI()
516
+ return _ai_service
517
+
518
+
519
+ def configure_ai_service(
520
+ groq_api_key: Optional[str] = None,
521
+ openrouter_api_key: Optional[str] = None,
522
+ openai_api_key: Optional[str] = None,
523
+ default_provider: AIProvider = AIProvider.GROQ,
524
+ ) -> MultiProviderAI:
525
+ """Configure and return the AI service with new settings."""
526
+ global _ai_service
527
+ _ai_service = MultiProviderAI(
528
+ groq_api_key=groq_api_key,
529
+ openrouter_api_key=openrouter_api_key,
530
+ openai_api_key=openai_api_key,
531
+ default_provider=default_provider,
532
+ )
533
+ return _ai_service
backend/app/ai/smart_router.py ADDED
@@ -0,0 +1,1097 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Smart AI Router with Auto-Swap
2
+ # Multi-provider AI routing with automatic failover for ContraVault
3
+ # Uses FREE tiers only: OpenRouter, Cerebras, Mistral
4
+
5
+ import os
6
+ import json
7
+ import httpx
8
+ import asyncio
9
+ import time
10
+ import logging
11
+ from typing import Optional, List, Dict, Any, AsyncGenerator, Callable
12
+ from pydantic import BaseModel, Field
13
+ from enum import Enum
14
+ from dataclasses import dataclass, field
15
+ from collections import defaultdict
16
+ from datetime import datetime, timedelta
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+
21
+ # =============================================================================
22
+ # ENUMS & CONSTANTS
23
+ # =============================================================================
24
+
25
+ class AIProvider(str, Enum):
26
+ """Supported AI providers (free tiers only)."""
27
+ OPENROUTER = "openrouter"
28
+ CEREBRAS = "cerebras"
29
+ MISTRAL = "mistral"
30
+
31
+
32
+ class FeatureType(str, Enum):
33
+ """Application features that require AI."""
34
+ DOCUMENT_INGESTION = "document_ingestion"
35
+ REQUIREMENT_SHREDDING = "requirement_shredding"
36
+ GO_NO_GO_ANALYSIS = "go_no_go_analysis"
37
+ RISK_ANALYSIS = "risk_analysis"
38
+ WIN_THEMES = "win_themes"
39
+ PROPOSAL_WRITING = "proposal_writing"
40
+ TEXT_IMPROVEMENTS = "text_improvements"
41
+ KNOWLEDGE_QA = "knowledge_qa"
42
+ CODE_GENERATION = "code_generation"
43
+ GENERAL = "general"
44
+
45
+
46
+ class ProviderHealth(str, Enum):
47
+ """Provider health states for circuit breaker."""
48
+ HEALTHY = "healthy"
49
+ DEGRADED = "degraded"
50
+ UNHEALTHY = "unhealthy"
51
+ RECOVERING = "recovering"
52
+
53
+
54
+ class SwapReason(str, Enum):
55
+ """Reasons for model swap."""
56
+ RATE_LIMIT = "rate_limit"
57
+ SERVER_ERROR = "server_error"
58
+ TIMEOUT = "timeout"
59
+ AUTH_ERROR = "auth_error"
60
+ CONTEXT_EXCEEDED = "context_exceeded"
61
+ INVALID_RESPONSE = "invalid_response"
62
+ PROVIDER_UNHEALTHY = "provider_unhealthy"
63
+
64
+
65
+ # =============================================================================
66
+ # MODELS & DATA CLASSES
67
+ # =============================================================================
68
+
69
+ class ModelConfig(BaseModel):
70
+ """Configuration for a specific AI model."""
71
+ provider: AIProvider
72
+ model_id: str
73
+ display_name: str
74
+ context_window: int
75
+ supports_streaming: bool = True
76
+ supports_vision: bool = False
77
+ is_free: bool = True
78
+
79
+
80
+ class ModelChain(BaseModel):
81
+ """Primary model with fallback chain."""
82
+ primary: ModelConfig
83
+ fallbacks: List[ModelConfig] = Field(default_factory=list)
84
+
85
+
86
+ class ChatMessage(BaseModel):
87
+ """Chat message format."""
88
+ role: str # "system", "user", "assistant"
89
+ content: str
90
+
91
+
92
+ class AIResponse(BaseModel):
93
+ """Standardized AI response."""
94
+ content: str
95
+ model: str
96
+ provider: AIProvider
97
+ usage: Optional[Dict[str, int]] = None
98
+ finish_reason: Optional[str] = None
99
+ swap_occurred: bool = False
100
+ swap_reason: Optional[SwapReason] = None
101
+ latency_ms: Optional[int] = None
102
+
103
+
104
+ @dataclass
105
+ class ProviderState:
106
+ """Tracks provider health state."""
107
+ health: ProviderHealth = ProviderHealth.HEALTHY
108
+ consecutive_failures: int = 0
109
+ last_failure_time: Optional[datetime] = None
110
+ last_success_time: Optional[datetime] = None
111
+ total_requests: int = 0
112
+ total_failures: int = 0
113
+ circuit_open_until: Optional[datetime] = None
114
+
115
+ @property
116
+ def error_rate(self) -> float:
117
+ if self.total_requests == 0:
118
+ return 0.0
119
+ return self.total_failures / self.total_requests
120
+
121
+
122
+ @dataclass
123
+ class RateLimitState:
124
+ """Tracks rate limit state per provider."""
125
+ requests_this_minute: int = 0
126
+ tokens_this_minute: int = 0
127
+ minute_start: datetime = field(default_factory=datetime.now)
128
+ backoff_until: Optional[datetime] = None
129
+
130
+
131
+ # =============================================================================
132
+ # FREE MODEL REGISTRY
133
+ # =============================================================================
134
+
135
+ # OpenRouter Free Models
136
+ OPENROUTER_MODELS = [
137
+ ModelConfig(
138
+ provider=AIProvider.OPENROUTER,
139
+ model_id="google/gemma-3-27b:free",
140
+ display_name="Gemma 3 27B",
141
+ context_window=131072,
142
+ supports_vision=True
143
+ ),
144
+ ModelConfig(
145
+ provider=AIProvider.OPENROUTER,
146
+ model_id="google/gemma-3-12b:free",
147
+ display_name="Gemma 3 12B",
148
+ context_window=32768,
149
+ supports_vision=True
150
+ ),
151
+ ModelConfig(
152
+ provider=AIProvider.OPENROUTER,
153
+ model_id="meta-llama/llama-3.3-70b-instruct:free",
154
+ display_name="Llama 3.3 70B",
155
+ context_window=131072
156
+ ),
157
+ ModelConfig(
158
+ provider=AIProvider.OPENROUTER,
159
+ model_id="meta-llama/llama-3.2-3b-instruct:free",
160
+ display_name="Llama 3.2 3B",
161
+ context_window=131072
162
+ ),
163
+ ModelConfig(
164
+ provider=AIProvider.OPENROUTER,
165
+ model_id="qwen/qwen2.5-vl-7b-instruct:free",
166
+ display_name="Qwen 2.5 VL 7B",
167
+ context_window=32768,
168
+ supports_vision=True
169
+ ),
170
+ ModelConfig(
171
+ provider=AIProvider.OPENROUTER,
172
+ model_id="nousresearch/hermes-3-llama-3.1-405b:free",
173
+ display_name="Hermes 3 405B",
174
+ context_window=131072
175
+ ),
176
+ ModelConfig(
177
+ provider=AIProvider.OPENROUTER,
178
+ model_id="meta-llama/llama-3.1-405b-instruct:free",
179
+ display_name="Llama 3.1 405B",
180
+ context_window=131072
181
+ ),
182
+ ]
183
+
184
+ # Cerebras Free Models
185
+ CEREBRAS_MODELS = [
186
+ ModelConfig(
187
+ provider=AIProvider.CEREBRAS,
188
+ model_id="llama-3.3-70b",
189
+ display_name="Llama 3.3 70B (Cerebras)",
190
+ context_window=65536
191
+ ),
192
+ ModelConfig(
193
+ provider=AIProvider.CEREBRAS,
194
+ model_id="llama3.1-8b",
195
+ display_name="Llama 3.1 8B (Cerebras)",
196
+ context_window=8192
197
+ ),
198
+ ModelConfig(
199
+ provider=AIProvider.CEREBRAS,
200
+ model_id="qwen-3-32b",
201
+ display_name="Qwen 3 32B (Cerebras)",
202
+ context_window=65536
203
+ ),
204
+ ModelConfig(
205
+ provider=AIProvider.CEREBRAS,
206
+ model_id="gpt-oss-120b",
207
+ display_name="GPT-OSS 120B (Cerebras)",
208
+ context_window=65536
209
+ ),
210
+ ]
211
+
212
+ # Mistral Free Models (Open-weight)
213
+ MISTRAL_MODELS = [
214
+ ModelConfig(
215
+ provider=AIProvider.MISTRAL,
216
+ model_id="open-mistral-nemo",
217
+ display_name="Mistral Nemo 12B",
218
+ context_window=131072
219
+ ),
220
+ ModelConfig(
221
+ provider=AIProvider.MISTRAL,
222
+ model_id="open-mixtral-8x7b",
223
+ display_name="Mixtral 8x7B",
224
+ context_window=32768
225
+ ),
226
+ ModelConfig(
227
+ provider=AIProvider.MISTRAL,
228
+ model_id="codestral-mamba",
229
+ display_name="Codestral Mamba",
230
+ context_window=262144
231
+ ),
232
+ ]
233
+
234
+ ALL_MODELS: Dict[AIProvider, List[ModelConfig]] = {
235
+ AIProvider.OPENROUTER: OPENROUTER_MODELS,
236
+ AIProvider.CEREBRAS: CEREBRAS_MODELS,
237
+ AIProvider.MISTRAL: MISTRAL_MODELS,
238
+ }
239
+
240
+
241
+ # =============================================================================
242
+ # FEATURE ROUTING CONFIGURATION
243
+ # =============================================================================
244
+
245
+ def get_model_by_id(model_id: str) -> Optional[ModelConfig]:
246
+ """Get model config by ID."""
247
+ for provider_models in ALL_MODELS.values():
248
+ for model in provider_models:
249
+ if model.model_id == model_id:
250
+ return model
251
+ return None
252
+
253
+
254
+ # Feature to Model Chain mapping
255
+ FEATURE_MODEL_CHAINS: Dict[FeatureType, ModelChain] = {
256
+ FeatureType.DOCUMENT_INGESTION: ModelChain(
257
+ primary=get_model_by_id("google/gemma-3-27b:free"),
258
+ fallbacks=[
259
+ get_model_by_id("llama-3.3-70b"),
260
+ get_model_by_id("meta-llama/llama-3.3-70b-instruct:free"),
261
+ ]
262
+ ),
263
+ FeatureType.REQUIREMENT_SHREDDING: ModelChain(
264
+ primary=get_model_by_id("meta-llama/llama-3.3-70b-instruct:free"),
265
+ fallbacks=[
266
+ get_model_by_id("qwen-3-32b"),
267
+ get_model_by_id("google/gemma-3-27b:free"),
268
+ ]
269
+ ),
270
+ FeatureType.GO_NO_GO_ANALYSIS: ModelChain(
271
+ primary=get_model_by_id("meta-llama/llama-3.1-405b-instruct:free"),
272
+ fallbacks=[
273
+ get_model_by_id("gpt-oss-120b"),
274
+ get_model_by_id("nousresearch/hermes-3-llama-3.1-405b:free"),
275
+ ]
276
+ ),
277
+ FeatureType.RISK_ANALYSIS: ModelChain(
278
+ primary=get_model_by_id("nousresearch/hermes-3-llama-3.1-405b:free"),
279
+ fallbacks=[
280
+ get_model_by_id("llama-3.3-70b"),
281
+ get_model_by_id("meta-llama/llama-3.3-70b-instruct:free"),
282
+ ]
283
+ ),
284
+ FeatureType.WIN_THEMES: ModelChain(
285
+ primary=get_model_by_id("google/gemma-3-27b:free"),
286
+ fallbacks=[
287
+ get_model_by_id("qwen-3-32b"),
288
+ get_model_by_id("meta-llama/llama-3.3-70b-instruct:free"),
289
+ ]
290
+ ),
291
+ FeatureType.PROPOSAL_WRITING: ModelChain(
292
+ primary=get_model_by_id("meta-llama/llama-3.1-405b-instruct:free"),
293
+ fallbacks=[
294
+ get_model_by_id("gpt-oss-120b"),
295
+ get_model_by_id("nousresearch/hermes-3-llama-3.1-405b:free"),
296
+ ]
297
+ ),
298
+ FeatureType.TEXT_IMPROVEMENTS: ModelChain(
299
+ primary=get_model_by_id("google/gemma-3-12b:free"),
300
+ fallbacks=[
301
+ get_model_by_id("llama3.1-8b"),
302
+ get_model_by_id("meta-llama/llama-3.2-3b-instruct:free"),
303
+ ]
304
+ ),
305
+ FeatureType.KNOWLEDGE_QA: ModelChain(
306
+ primary=get_model_by_id("qwen-3-32b"),
307
+ fallbacks=[
308
+ get_model_by_id("meta-llama/llama-3.3-70b-instruct:free"),
309
+ get_model_by_id("google/gemma-3-27b:free"),
310
+ ]
311
+ ),
312
+ FeatureType.CODE_GENERATION: ModelChain(
313
+ primary=get_model_by_id("codestral-mamba"),
314
+ fallbacks=[
315
+ get_model_by_id("meta-llama/llama-3.3-70b-instruct:free"),
316
+ get_model_by_id("qwen-3-32b"),
317
+ ]
318
+ ),
319
+ FeatureType.GENERAL: ModelChain(
320
+ primary=get_model_by_id("meta-llama/llama-3.3-70b-instruct:free"),
321
+ fallbacks=[
322
+ get_model_by_id("llama-3.3-70b"),
323
+ get_model_by_id("google/gemma-3-27b:free"),
324
+ ]
325
+ ),
326
+ }
327
+
328
+
329
+ # =============================================================================
330
+ # PROVIDER ADAPTERS
331
+ # =============================================================================
332
+
333
+ class ProviderAdapter:
334
+ """Base adapter for AI providers."""
335
+
336
+ def __init__(self, api_key: str, base_url: str):
337
+ self.api_key = api_key
338
+ self.base_url = base_url
339
+
340
+ def get_headers(self) -> Dict[str, str]:
341
+ return {
342
+ "Content-Type": "application/json",
343
+ "Authorization": f"Bearer {self.api_key}",
344
+ }
345
+
346
+ async def complete(
347
+ self,
348
+ messages: List[ChatMessage],
349
+ model: str,
350
+ temperature: float = 0.7,
351
+ max_tokens: int = 4096,
352
+ ) -> Dict[str, Any]:
353
+ raise NotImplementedError
354
+
355
+
356
+ class OpenRouterAdapter(ProviderAdapter):
357
+ """Adapter for OpenRouter API."""
358
+
359
+ def __init__(self, api_key: str):
360
+ super().__init__(api_key, "https://openrouter.ai/api/v1")
361
+
362
+ def get_headers(self) -> Dict[str, str]:
363
+ headers = super().get_headers()
364
+ headers["HTTP-Referer"] = "https://contravault.ai"
365
+ headers["X-Title"] = "ContraVault RFP Platform"
366
+ return headers
367
+
368
+ async def complete(
369
+ self,
370
+ messages: List[ChatMessage],
371
+ model: str,
372
+ temperature: float = 0.7,
373
+ max_tokens: int = 4096,
374
+ ) -> Dict[str, Any]:
375
+ url = f"{self.base_url}/chat/completions"
376
+ payload = {
377
+ "model": model,
378
+ "messages": [{"role": m.role, "content": m.content} for m in messages],
379
+ "temperature": temperature,
380
+ "max_tokens": max_tokens,
381
+ }
382
+
383
+ async with httpx.AsyncClient(timeout=120.0) as client:
384
+ response = await client.post(url, headers=self.get_headers(), json=payload)
385
+ response.raise_for_status()
386
+ return response.json()
387
+
388
+
389
+ class CerebrasAdapter(ProviderAdapter):
390
+ """Adapter for Cerebras API."""
391
+
392
+ def __init__(self, api_key: str):
393
+ super().__init__(api_key, "https://api.cerebras.ai/v1")
394
+
395
+ async def complete(
396
+ self,
397
+ messages: List[ChatMessage],
398
+ model: str,
399
+ temperature: float = 0.7,
400
+ max_tokens: int = 4096,
401
+ ) -> Dict[str, Any]:
402
+ url = f"{self.base_url}/chat/completions"
403
+ payload = {
404
+ "model": model,
405
+ "messages": [{"role": m.role, "content": m.content} for m in messages],
406
+ "temperature": temperature,
407
+ "max_tokens": max_tokens,
408
+ }
409
+
410
+ async with httpx.AsyncClient(timeout=120.0) as client:
411
+ response = await client.post(url, headers=self.get_headers(), json=payload)
412
+ response.raise_for_status()
413
+ return response.json()
414
+
415
+
416
+ class MistralAdapter(ProviderAdapter):
417
+ """Adapter for Mistral API."""
418
+
419
+ def __init__(self, api_key: str):
420
+ super().__init__(api_key, "https://api.mistral.ai/v1")
421
+
422
+ async def complete(
423
+ self,
424
+ messages: List[ChatMessage],
425
+ model: str,
426
+ temperature: float = 0.7,
427
+ max_tokens: int = 4096,
428
+ ) -> Dict[str, Any]:
429
+ url = f"{self.base_url}/chat/completions"
430
+ payload = {
431
+ "model": model,
432
+ "messages": [{"role": m.role, "content": m.content} for m in messages],
433
+ "temperature": temperature,
434
+ "max_tokens": max_tokens,
435
+ }
436
+
437
+ async with httpx.AsyncClient(timeout=120.0) as client:
438
+ response = await client.post(url, headers=self.get_headers(), json=payload)
439
+ response.raise_for_status()
440
+ return response.json()
441
+
442
+
443
+ # =============================================================================
444
+ # HEALTH MONITOR
445
+ # =============================================================================
446
+
447
+ class HealthMonitor:
448
+ """Monitors provider health with circuit breaker pattern."""
449
+
450
+ FAILURE_THRESHOLD = 5
451
+ RECOVERY_TIMEOUT = 60 # seconds
452
+ DEGRADED_THRESHOLD = 0.2 # 20% error rate
453
+
454
+ def __init__(self):
455
+ self.provider_states: Dict[AIProvider, ProviderState] = {
456
+ provider: ProviderState() for provider in AIProvider
457
+ }
458
+ self.rate_limits: Dict[AIProvider, RateLimitState] = {
459
+ provider: RateLimitState() for provider in AIProvider
460
+ }
461
+
462
+ def is_healthy(self, provider: AIProvider) -> bool:
463
+ """Check if provider is available for requests."""
464
+ state = self.provider_states[provider]
465
+
466
+ # Check circuit breaker
467
+ if state.circuit_open_until:
468
+ if datetime.now() < state.circuit_open_until:
469
+ return False
470
+ else:
471
+ # Circuit recovery period
472
+ state.health = ProviderHealth.RECOVERING
473
+ state.circuit_open_until = None
474
+
475
+ return state.health in [ProviderHealth.HEALTHY, ProviderHealth.DEGRADED, ProviderHealth.RECOVERING]
476
+
477
+ def record_success(self, provider: AIProvider):
478
+ """Record successful request."""
479
+ state = self.provider_states[provider]
480
+ state.total_requests += 1
481
+ state.last_success_time = datetime.now()
482
+ state.consecutive_failures = 0
483
+
484
+ # Heal degraded state
485
+ if state.health in [ProviderHealth.DEGRADED, ProviderHealth.RECOVERING]:
486
+ if state.error_rate < self.DEGRADED_THRESHOLD:
487
+ state.health = ProviderHealth.HEALTHY
488
+
489
+ logger.debug(f"Provider {provider.value} success recorded")
490
+
491
+ def record_failure(self, provider: AIProvider, reason: SwapReason):
492
+ """Record failed request and potentially open circuit."""
493
+ state = self.provider_states[provider]
494
+ state.total_requests += 1
495
+ state.total_failures += 1
496
+ state.consecutive_failures += 1
497
+ state.last_failure_time = datetime.now()
498
+
499
+ # Check for circuit breaker trigger
500
+ if state.consecutive_failures >= self.FAILURE_THRESHOLD:
501
+ state.health = ProviderHealth.UNHEALTHY
502
+ state.circuit_open_until = datetime.now() + timedelta(seconds=self.RECOVERY_TIMEOUT)
503
+ logger.warning(
504
+ f"Circuit breaker OPEN for {provider.value} after {state.consecutive_failures} failures. "
505
+ f"Will recover at {state.circuit_open_until}"
506
+ )
507
+ elif state.error_rate > self.DEGRADED_THRESHOLD:
508
+ state.health = ProviderHealth.DEGRADED
509
+ logger.warning(f"Provider {provider.value} degraded (error rate: {state.error_rate:.1%})")
510
+
511
+ logger.warning(f"Provider {provider.value} failure: {reason.value}")
512
+
513
+ def check_rate_limit(self, provider: AIProvider) -> bool:
514
+ """Check if rate limit allows request."""
515
+ rate_state = self.rate_limits[provider]
516
+
517
+ # Check backoff
518
+ if rate_state.backoff_until and datetime.now() < rate_state.backoff_until:
519
+ return False
520
+
521
+ # Reset minute counter if needed
522
+ if datetime.now() - rate_state.minute_start > timedelta(minutes=1):
523
+ rate_state.requests_this_minute = 0
524
+ rate_state.tokens_this_minute = 0
525
+ rate_state.minute_start = datetime.now()
526
+ rate_state.backoff_until = None
527
+
528
+ # Check limits (conservative estimates for free tiers)
529
+ limits = {
530
+ AIProvider.OPENROUTER: 60,
531
+ AIProvider.CEREBRAS: 30,
532
+ AIProvider.MISTRAL: 60,
533
+ }
534
+
535
+ return rate_state.requests_this_minute < limits.get(provider, 30)
536
+
537
+ def record_rate_limit_hit(self, provider: AIProvider):
538
+ """Record rate limit hit and set backoff."""
539
+ rate_state = self.rate_limits[provider]
540
+
541
+ # Exponential backoff: 5s, 10s, 20s, 40s, max 300s
542
+ current_backoff = 5
543
+ if rate_state.backoff_until:
544
+ time_since_backoff = (datetime.now() - rate_state.backoff_until).total_seconds()
545
+ if time_since_backoff < 60: # Recent backoff
546
+ current_backoff = min(current_backoff * 2, 300)
547
+
548
+ rate_state.backoff_until = datetime.now() + timedelta(seconds=current_backoff)
549
+ logger.warning(f"Rate limit hit for {provider.value}, backing off for {current_backoff}s")
550
+
551
+ def get_status(self) -> Dict[str, Any]:
552
+ """Get health status for all providers."""
553
+ return {
554
+ provider.value: {
555
+ "health": self.provider_states[provider].health.value,
556
+ "error_rate": f"{self.provider_states[provider].error_rate:.1%}",
557
+ "consecutive_failures": self.provider_states[provider].consecutive_failures,
558
+ "total_requests": self.provider_states[provider].total_requests,
559
+ "rate_limited": not self.check_rate_limit(provider),
560
+ }
561
+ for provider in AIProvider
562
+ }
563
+
564
+
565
+ # =============================================================================
566
+ # SMART AI ROUTER
567
+ # =============================================================================
568
+
569
+ class SmartAIRouter:
570
+ """
571
+ Smart AI Router with automatic failover.
572
+ Routes requests to optimal free-tier models based on feature type.
573
+ Automatically swaps providers on failure or rate limits.
574
+ """
575
+
576
+ def __init__(
577
+ self,
578
+ openrouter_api_key: Optional[str] = None,
579
+ cerebras_api_key: Optional[str] = None,
580
+ mistral_api_key: Optional[str] = None,
581
+ ):
582
+ # Load API keys from environment if not provided
583
+ self.api_keys = {
584
+ AIProvider.OPENROUTER: openrouter_api_key or os.getenv("OPENROUTER_API_KEY", ""),
585
+ AIProvider.CEREBRAS: cerebras_api_key or os.getenv("CEREBRAS_API_KEY", ""),
586
+ AIProvider.MISTRAL: mistral_api_key or os.getenv("MISTRAL_API_KEY", ""),
587
+ }
588
+
589
+ # Initialize adapters
590
+ self.adapters: Dict[AIProvider, Optional[ProviderAdapter]] = {}
591
+ self._init_adapters()
592
+
593
+ # Health monitor
594
+ self.health_monitor = HealthMonitor()
595
+
596
+ # Swap history for debugging
597
+ self.swap_history: List[Dict[str, Any]] = []
598
+
599
+ def _init_adapters(self):
600
+ """Initialize provider adapters."""
601
+ if self.api_keys[AIProvider.OPENROUTER]:
602
+ self.adapters[AIProvider.OPENROUTER] = OpenRouterAdapter(
603
+ self.api_keys[AIProvider.OPENROUTER]
604
+ )
605
+
606
+ if self.api_keys[AIProvider.CEREBRAS]:
607
+ self.adapters[AIProvider.CEREBRAS] = CerebrasAdapter(
608
+ self.api_keys[AIProvider.CEREBRAS]
609
+ )
610
+
611
+ if self.api_keys[AIProvider.MISTRAL]:
612
+ self.adapters[AIProvider.MISTRAL] = MistralAdapter(
613
+ self.api_keys[AIProvider.MISTRAL]
614
+ )
615
+
616
+ def _get_model_chain(self, feature: FeatureType) -> List[ModelConfig]:
617
+ """Get ordered list of models to try for a feature."""
618
+ chain = FEATURE_MODEL_CHAINS.get(feature, FEATURE_MODEL_CHAINS[FeatureType.GENERAL])
619
+ models = [chain.primary] + chain.fallbacks
620
+
621
+ # Filter out None models and providers without API keys
622
+ return [
623
+ m for m in models
624
+ if m is not None and self.api_keys.get(m.provider)
625
+ ]
626
+
627
+ def _determine_swap_reason(self, error: Exception) -> SwapReason:
628
+ """Determine the reason for swap based on error."""
629
+ error_str = str(error).lower()
630
+
631
+ if isinstance(error, httpx.HTTPStatusError):
632
+ status = error.response.status_code
633
+ if status == 429:
634
+ return SwapReason.RATE_LIMIT
635
+ elif status in [401, 403]:
636
+ return SwapReason.AUTH_ERROR
637
+ elif status >= 500:
638
+ return SwapReason.SERVER_ERROR
639
+ elif status == 400 and "context" in error_str:
640
+ return SwapReason.CONTEXT_EXCEEDED
641
+
642
+ if isinstance(error, httpx.TimeoutException):
643
+ return SwapReason.TIMEOUT
644
+
645
+ if "json" in error_str or "parse" in error_str:
646
+ return SwapReason.INVALID_RESPONSE
647
+
648
+ return SwapReason.SERVER_ERROR
649
+
650
+ def _log_swap(
651
+ self,
652
+ from_model: ModelConfig,
653
+ to_model: Optional[ModelConfig],
654
+ reason: SwapReason,
655
+ feature: FeatureType,
656
+ ):
657
+ """Log model swap for debugging."""
658
+ swap_record = {
659
+ "timestamp": datetime.now().isoformat(),
660
+ "feature": feature.value,
661
+ "from_provider": from_model.provider.value,
662
+ "from_model": from_model.model_id,
663
+ "to_provider": to_model.provider.value if to_model else None,
664
+ "to_model": to_model.model_id if to_model else None,
665
+ "reason": reason.value,
666
+ }
667
+ self.swap_history.append(swap_record)
668
+
669
+ # Keep only last 100 swaps
670
+ if len(self.swap_history) > 100:
671
+ self.swap_history = self.swap_history[-100:]
672
+
673
+ logger.info(
674
+ f"AI SWAP: {from_model.display_name} -> {to_model.display_name if to_model else 'NONE'} "
675
+ f"[{reason.value}] for {feature.value}"
676
+ )
677
+
678
+ async def complete(
679
+ self,
680
+ messages: List[ChatMessage],
681
+ feature: FeatureType = FeatureType.GENERAL,
682
+ temperature: float = 0.7,
683
+ max_tokens: int = 4096,
684
+ force_model: Optional[str] = None,
685
+ force_provider: Optional[AIProvider] = None,
686
+ ) -> AIResponse:
687
+ """
688
+ Complete a chat request with automatic failover.
689
+
690
+ Args:
691
+ messages: Chat messages
692
+ feature: Feature type for optimal model selection
693
+ temperature: Sampling temperature
694
+ max_tokens: Maximum tokens in response
695
+ force_model: Force specific model (bypasses feature routing)
696
+ force_provider: Force specific provider
697
+
698
+ Returns:
699
+ AIResponse with content and metadata
700
+ """
701
+ start_time = time.time()
702
+
703
+ # Get model chain for feature
704
+ if force_model and force_provider:
705
+ model_chain = [get_model_by_id(force_model)]
706
+ if not model_chain[0]:
707
+ model_chain = self._get_model_chain(feature)
708
+ else:
709
+ model_chain = self._get_model_chain(feature)
710
+
711
+ if not model_chain:
712
+ raise ValueError(f"No models available for feature {feature.value}")
713
+
714
+ swap_occurred = False
715
+ swap_reason = None
716
+ last_error = None
717
+ attempted_models = []
718
+
719
+ for i, model in enumerate(model_chain):
720
+ if model is None:
721
+ continue
722
+
723
+ attempted_models.append(model.display_name)
724
+
725
+ # Check provider health
726
+ if not self.health_monitor.is_healthy(model.provider):
727
+ logger.debug(f"Skipping unhealthy provider {model.provider.value}")
728
+ if i > 0:
729
+ swap_occurred = True
730
+ swap_reason = SwapReason.PROVIDER_UNHEALTHY
731
+ continue
732
+
733
+ # Check rate limit
734
+ if not self.health_monitor.check_rate_limit(model.provider):
735
+ logger.debug(f"Skipping rate-limited provider {model.provider.value}")
736
+ if i > 0:
737
+ swap_occurred = True
738
+ swap_reason = SwapReason.RATE_LIMIT
739
+ continue
740
+
741
+ adapter = self.adapters.get(model.provider)
742
+ if not adapter:
743
+ continue
744
+
745
+ try:
746
+ logger.debug(f"Trying {model.display_name} ({model.provider.value})")
747
+
748
+ response_data = await adapter.complete(
749
+ messages=messages,
750
+ model=model.model_id,
751
+ temperature=temperature,
752
+ max_tokens=max_tokens,
753
+ )
754
+
755
+ # Parse response
756
+ choice = response_data["choices"][0]
757
+ content = choice["message"]["content"]
758
+
759
+ if not content or not content.strip():
760
+ raise ValueError("Empty response content")
761
+
762
+ # Record success
763
+ self.health_monitor.record_success(model.provider)
764
+
765
+ latency_ms = int((time.time() - start_time) * 1000)
766
+
767
+ return AIResponse(
768
+ content=content,
769
+ model=response_data.get("model", model.model_id),
770
+ provider=model.provider,
771
+ usage=response_data.get("usage"),
772
+ finish_reason=choice.get("finish_reason"),
773
+ swap_occurred=swap_occurred,
774
+ swap_reason=swap_reason,
775
+ latency_ms=latency_ms,
776
+ )
777
+
778
+ except Exception as e:
779
+ last_error = e
780
+ reason = self._determine_swap_reason(e)
781
+
782
+ # Record failure
783
+ self.health_monitor.record_failure(model.provider, reason)
784
+
785
+ if reason == SwapReason.RATE_LIMIT:
786
+ self.health_monitor.record_rate_limit_hit(model.provider)
787
+
788
+ # Log swap if moving to next model
789
+ if i < len(model_chain) - 1:
790
+ next_model = model_chain[i + 1] if i + 1 < len(model_chain) else None
791
+ self._log_swap(model, next_model, reason, feature)
792
+ swap_occurred = True
793
+ swap_reason = reason
794
+
795
+ logger.warning(f"Model {model.display_name} failed: {e}")
796
+ continue
797
+
798
+ # All models exhausted
799
+ raise Exception(
800
+ f"All models exhausted for {feature.value}. "
801
+ f"Attempted: {', '.join(attempted_models)}. "
802
+ f"Last error: {last_error}"
803
+ )
804
+
805
+ async def analyze_rfp(
806
+ self,
807
+ text: str,
808
+ analysis_type: str = "requirements",
809
+ ) -> Dict[str, Any]:
810
+ """
811
+ Specialized RFP analysis using optimal models.
812
+
813
+ Args:
814
+ text: RFP text to analyze
815
+ analysis_type: Type of analysis (requirements, risks, go_no_go, etc.)
816
+
817
+ Returns:
818
+ Structured analysis results
819
+ """
820
+ # Map analysis type to feature
821
+ feature_map = {
822
+ "requirements": FeatureType.REQUIREMENT_SHREDDING,
823
+ "risks": FeatureType.RISK_ANALYSIS,
824
+ "go_no_go": FeatureType.GO_NO_GO_ANALYSIS,
825
+ "win_themes": FeatureType.WIN_THEMES,
826
+ "compliance": FeatureType.REQUIREMENT_SHREDDING,
827
+ }
828
+
829
+ feature = feature_map.get(analysis_type, FeatureType.GENERAL)
830
+
831
+ prompts = {
832
+ "requirements": """Analyze the following RFP text and extract all requirements.
833
+ For each requirement, identify:
834
+ 1. The exact text of the requirement
835
+ 2. Whether it's mandatory (shall/must) or desirable (should/may)
836
+ 3. The category (Technical, Management, Legal, Financial, etc.)
837
+ 4. Risk level (High/Medium/Low)
838
+ 5. Any keywords or compliance standards mentioned
839
+
840
+ Return as JSON array with fields: text, type, category, risk_level, keywords
841
+
842
+ RFP Text:
843
+ {text}""",
844
+
845
+ "risks": """Analyze the following RFP text for potential risks and red flags.
846
+ Look for:
847
+ 1. Unlimited liability clauses
848
+ 2. Unrealistic timelines
849
+ 3. Ambiguous scope
850
+ 4. Onerous payment terms
851
+ 5. IP ownership issues
852
+ 6. Performance bond requirements
853
+ 7. Liquidated damages
854
+ 8. Insurance requirements
855
+
856
+ Return as JSON array with fields: risk_type, severity (high/medium/low), clause_text, recommendation
857
+
858
+ RFP Text:
859
+ {text}""",
860
+
861
+ "go_no_go": """Perform a Go/No-Go analysis on this RFP.
862
+ Evaluate:
863
+ 1. Technical fit (1-10)
864
+ 2. Resource availability (1-10)
865
+ 3. Timeline feasibility (1-10)
866
+ 4. Competitive position (1-10)
867
+ 5. Profitability potential (1-10)
868
+ 6. Strategic alignment (1-10)
869
+ 7. Risk level (1-10, lower is better)
870
+
871
+ Provide:
872
+ - Overall recommendation: GO, NO-GO, or CONDITIONAL
873
+ - Confidence percentage
874
+ - Key reasons for the decision
875
+ - Mitigation strategies if CONDITIONAL
876
+
877
+ Return as JSON with fields: recommendation, confidence, scores (object), reasons (array), mitigations (array)
878
+
879
+ RFP Text:
880
+ {text}""",
881
+
882
+ "win_themes": """Based on this RFP, identify potential win themes and discriminators.
883
+ Consider:
884
+ 1. What does the client really need?
885
+ 2. What pain points are implied?
886
+ 3. What differentiators could be emphasized?
887
+ 4. What proof points would be compelling?
888
+
889
+ Return as JSON array with fields: theme_title, description, supporting_evidence, discriminator_level (strong/moderate/weak)
890
+
891
+ RFP Text:
892
+ {text}""",
893
+
894
+ "compliance": """Analyze this RFP for compliance requirements.
895
+ Identify:
896
+ 1. Certifications required (ISO, SOC2, FedRAMP, etc.)
897
+ 2. Security standards
898
+ 3. Regulatory compliance (GDPR, HIPAA, etc.)
899
+ 4. Format requirements (page limits, fonts, sections)
900
+ 5. Submission requirements (deadlines, methods)
901
+
902
+ Return as JSON with fields: certifications (array), security_standards (array), regulations (array), format_requirements (object), submission_requirements (object)
903
+
904
+ RFP Text:
905
+ {text}""",
906
+ }
907
+
908
+ prompt = prompts.get(analysis_type, prompts["requirements"])
909
+ formatted_prompt = prompt.format(text=text[:15000])
910
+
911
+ messages = [
912
+ ChatMessage(
913
+ role="system",
914
+ content="You are an expert RFP analyst. Always respond with valid JSON only, no markdown or explanations."
915
+ ),
916
+ ChatMessage(role="user", content=formatted_prompt),
917
+ ]
918
+
919
+ response = await self.complete(messages, feature=feature, temperature=0.3)
920
+
921
+ # Parse JSON response
922
+ try:
923
+ content = response.content.strip()
924
+ if content.startswith("```"):
925
+ content = content.split("```")[1]
926
+ if content.startswith("json"):
927
+ content = content[4:]
928
+ return json.loads(content)
929
+ except json.JSONDecodeError:
930
+ return {"raw_response": response.content, "error": "Failed to parse JSON response"}
931
+
932
+ async def generate_proposal_section(
933
+ self,
934
+ section_name: str,
935
+ requirements: List[str],
936
+ company_info: Dict[str, Any],
937
+ style_guide: Optional[str] = None,
938
+ ) -> str:
939
+ """Generate a proposal section using optimal model."""
940
+ system_prompt = f"""You are an expert proposal writer for government and enterprise RFPs.
941
+ Write professional, compliant proposal content that:
942
+ 1. Directly addresses each requirement
943
+ 2. Uses clear, concise language
944
+ 3. Highlights the company's strengths and experience
945
+ 4. Follows any provided style guidelines
946
+ 5. Maintains a confident but not arrogant tone
947
+
948
+ Company Information:
949
+ {json.dumps(company_info, indent=2)}
950
+
951
+ {"Style Guide: " + style_guide if style_guide else ""}
952
+ """
953
+
954
+ user_prompt = f"""Write the "{section_name}" section for a proposal.
955
+
956
+ Requirements to address:
957
+ {chr(10).join(f"- {r}" for r in requirements)}
958
+
959
+ Generate professional proposal content that addresses each requirement. Include specific details and evidence where appropriate.
960
+ """
961
+
962
+ messages = [
963
+ ChatMessage(role="system", content=system_prompt),
964
+ ChatMessage(role="user", content=user_prompt),
965
+ ]
966
+
967
+ response = await self.complete(
968
+ messages,
969
+ feature=FeatureType.PROPOSAL_WRITING,
970
+ temperature=0.7,
971
+ max_tokens=4096
972
+ )
973
+ return response.content
974
+
975
+ async def suggest_improvements(
976
+ self,
977
+ text: str,
978
+ context: str = "proposal",
979
+ ) -> List[Dict[str, str]]:
980
+ """Suggest improvements for proposal text."""
981
+ prompt = f"""Review the following {context} text and suggest improvements.
982
+ For each suggestion, provide:
983
+ 1. The original text segment
984
+ 2. The suggested improvement
985
+ 3. The reason for the change
986
+ 4. Priority (high/medium/low)
987
+
988
+ Focus on:
989
+ - Clarity and conciseness
990
+ - Compliance with typical RFP requirements
991
+ - Professional tone
992
+ - Specificity and evidence
993
+
994
+ Text to review:
995
+ {text}
996
+
997
+ Return as JSON array with fields: original, suggestion, reason, priority
998
+ """
999
+
1000
+ messages = [
1001
+ ChatMessage(
1002
+ role="system",
1003
+ content="You are an expert proposal editor. Always respond with valid JSON only."
1004
+ ),
1005
+ ChatMessage(role="user", content=prompt),
1006
+ ]
1007
+
1008
+ response = await self.complete(
1009
+ messages,
1010
+ feature=FeatureType.TEXT_IMPROVEMENTS,
1011
+ temperature=0.5
1012
+ )
1013
+
1014
+ try:
1015
+ content = response.content.strip()
1016
+ if content.startswith("```"):
1017
+ content = content.split("```")[1]
1018
+ if content.startswith("json"):
1019
+ content = content[4:]
1020
+ return json.loads(content)
1021
+ except json.JSONDecodeError:
1022
+ return [{"error": "Failed to parse suggestions", "raw": response.content}]
1023
+
1024
+ def get_available_providers(self) -> List[Dict[str, Any]]:
1025
+ """Get list of available providers with their status."""
1026
+ providers = []
1027
+ for provider in AIProvider:
1028
+ has_key = bool(self.api_keys.get(provider))
1029
+ models = ALL_MODELS.get(provider, [])
1030
+ health_status = self.health_monitor.get_status().get(provider.value, {})
1031
+
1032
+ providers.append({
1033
+ "id": provider.value,
1034
+ "name": provider.value.title(),
1035
+ "configured": has_key,
1036
+ "health": health_status,
1037
+ "models": [m.model_dump() for m in models],
1038
+ })
1039
+ return providers
1040
+
1041
+ def get_available_models(self, provider: Optional[AIProvider] = None) -> List[ModelConfig]:
1042
+ """Get available models for a specific provider or all providers."""
1043
+ if provider:
1044
+ return ALL_MODELS.get(provider, [])
1045
+
1046
+ all_models = []
1047
+ for p, models in ALL_MODELS.items():
1048
+ if self.api_keys.get(p):
1049
+ all_models.extend(models)
1050
+ return all_models
1051
+
1052
+ def get_health_status(self) -> Dict[str, Any]:
1053
+ """Get health status for all providers."""
1054
+ return {
1055
+ "providers": self.health_monitor.get_status(),
1056
+ "recent_swaps": self.swap_history[-10:] if self.swap_history else [],
1057
+ }
1058
+
1059
+
1060
+ # =============================================================================
1061
+ # SINGLETON INSTANCE
1062
+ # =============================================================================
1063
+
1064
+ _smart_router: Optional[SmartAIRouter] = None
1065
+
1066
+
1067
+ def get_smart_router() -> SmartAIRouter:
1068
+ """Get or create the Smart AI Router singleton."""
1069
+ global _smart_router
1070
+ if _smart_router is None:
1071
+ _smart_router = SmartAIRouter()
1072
+ return _smart_router
1073
+
1074
+
1075
+ def configure_smart_router(
1076
+ openrouter_api_key: Optional[str] = None,
1077
+ cerebras_api_key: Optional[str] = None,
1078
+ mistral_api_key: Optional[str] = None,
1079
+ ) -> SmartAIRouter:
1080
+ """Configure and return the Smart AI Router with new settings."""
1081
+ global _smart_router
1082
+ _smart_router = SmartAIRouter(
1083
+ openrouter_api_key=openrouter_api_key,
1084
+ cerebras_api_key=cerebras_api_key,
1085
+ mistral_api_key=mistral_api_key,
1086
+ )
1087
+ return _smart_router
1088
+
1089
+
1090
+ # =============================================================================
1091
+ # BACKWARDS COMPATIBILITY
1092
+ # =============================================================================
1093
+
1094
+ # Alias for backwards compatibility with existing code
1095
+ MultiProviderAI = SmartAIRouter
1096
+ get_ai_service = get_smart_router
1097
+ configure_ai_service = configure_smart_router
backend/app/analysis/__init__.py ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Analysis Engine module
2
+ from .service import AnalysisEngine
3
+ from .models import GoNoGoAnalysis, RiskAssessment, DecisionEnum, RiskLevel, ComplianceStatus
4
+ from .competitor_db import CompetitorDatabase, Competitor, CompetitorCapability
5
+ from .nli_detector import NLIContradictionDetector
6
+
7
+ __all__ = [
8
+ "AnalysisEngine",
9
+ "GoNoGoAnalysis",
10
+ "RiskAssessment",
11
+ "DecisionEnum",
12
+ "RiskLevel",
13
+ "ComplianceStatus",
14
+ "CompetitorDatabase",
15
+ "Competitor",
16
+ "CompetitorCapability",
17
+ "NLIContradictionDetector"
18
+ ]
backend/app/analysis/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (601 Bytes). View file
 
backend/app/analysis/__pycache__/competitor_db.cpython-313.pyc ADDED
Binary file (22.3 kB). View file
 
backend/app/analysis/__pycache__/models.cpython-313.pyc ADDED
Binary file (4.25 kB). View file
 
backend/app/analysis/__pycache__/nli_detector.cpython-313.pyc ADDED
Binary file (22.3 kB). View file
 
backend/app/analysis/__pycache__/service.cpython-313.pyc ADDED
Binary file (39.7 kB). View file
 
backend/app/analysis/competitor_db.py ADDED
@@ -0,0 +1,502 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Competitor Database Module - Semantic Blind RFP Detection
3
+
4
+ Implements:
5
+ - 4.2: Semantic "Blind RFP" Detection
6
+ - Competitor capability database management
7
+ - Semantic similarity comparison for wired RFP detection
8
+ """
9
+
10
+ import logging
11
+ from typing import List, Dict, Any, Optional, Tuple
12
+ from dataclasses import dataclass, field, asdict
13
+ import json
14
+ import os
15
+ from datetime import datetime
16
+ import re
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+
21
+ @dataclass
22
+ class CompetitorCapability:
23
+ """Represents a competitor's specific capability or product."""
24
+ name: str
25
+ description: str
26
+ proprietary: bool = False # Is this unique to this competitor?
27
+ keywords: List[str] = field(default_factory=list)
28
+ certifications: List[str] = field(default_factory=list)
29
+
30
+
31
+ @dataclass
32
+ class Competitor:
33
+ """Represents a competitor company profile."""
34
+ id: str
35
+ name: str
36
+ description: str
37
+ capabilities: List[CompetitorCapability] = field(default_factory=list)
38
+ proprietary_technologies: List[str] = field(default_factory=list)
39
+ certifications: List[str] = field(default_factory=list)
40
+ contract_vehicles: List[str] = field(default_factory=list)
41
+ past_performance_agencies: List[str] = field(default_factory=list)
42
+ strengths: List[str] = field(default_factory=list)
43
+ weaknesses: List[str] = field(default_factory=list)
44
+ incumbent_contracts: List[str] = field(default_factory=list)
45
+
46
+ def to_dict(self) -> Dict[str, Any]:
47
+ result = asdict(self)
48
+ result['capabilities'] = [asdict(c) for c in self.capabilities]
49
+ return result
50
+
51
+
52
+ @dataclass
53
+ class BlindRFPMatch:
54
+ """Represents a potential match indicating a wired/blind RFP."""
55
+ competitor_id: str
56
+ competitor_name: str
57
+ match_type: str # "proprietary_tech", "capability", "incumbent", "certification"
58
+ requirement_text: str
59
+ matched_element: str
60
+ confidence: float
61
+ explanation: str
62
+
63
+
64
+ class CompetitorDatabase:
65
+ """
66
+ Manages competitor intelligence for blind RFP detection.
67
+
68
+ Features:
69
+ - Store competitor profiles with capabilities
70
+ - Semantic similarity matching between RFP requirements and competitor capabilities
71
+ - Detect proprietary technology requirements
72
+ - Incumbent advantage detection
73
+ """
74
+
75
+ def __init__(self, db_path: Optional[str] = None):
76
+ """
77
+ Initialize competitor database.
78
+
79
+ Args:
80
+ db_path: Path to JSON file for persistence
81
+ """
82
+ self.db_path = db_path or "./data/competitors.json"
83
+ self.competitors: Dict[str, Competitor] = {}
84
+ self._load_database()
85
+
86
+ # Common proprietary technology patterns
87
+ self.proprietary_patterns = {
88
+ "ServiceNow": ["servicenow", "snow platform", "now platform"],
89
+ "Salesforce": ["salesforce", "force.com", "apex", "lightning"],
90
+ "SAP": ["sap s/4hana", "sap erp", "sap hana"],
91
+ "Oracle": ["oracle cloud", "oracle fusion", "oracle ebusiness"],
92
+ "Microsoft": ["dynamics 365", "azure government", "m365 gcc"],
93
+ "AWS": ["aws govcloud", "amazon connect"],
94
+ "Palantir": ["palantir foundry", "gotham"],
95
+ "Splunk": ["splunk enterprise", "splunk cloud"],
96
+ "Pega": ["pegasystems", "pega platform"],
97
+ }
98
+
99
+ # Contract vehicle patterns
100
+ self.contract_vehicles = {
101
+ "GSA MAS": ["gsa schedule", "gsa mas", "federal supply schedule"],
102
+ "OASIS": ["oasis", "oasis sb", "oasis unrestricted"],
103
+ "SEWP": ["sewp v", "nasa sewp"],
104
+ "CIO-SP3": ["cio-sp3", "cio sp3"],
105
+ "Alliant": ["alliant 2", "alliant ii"],
106
+ "8(a) STARS": ["8a stars", "stars iii"],
107
+ }
108
+
109
+ def _load_database(self):
110
+ """Load competitor data from file."""
111
+ if os.path.exists(self.db_path):
112
+ try:
113
+ with open(self.db_path, 'r') as f:
114
+ data = json.load(f)
115
+ for comp_data in data.get('competitors', []):
116
+ capabilities = [
117
+ CompetitorCapability(**cap)
118
+ for cap in comp_data.pop('capabilities', [])
119
+ ]
120
+ self.competitors[comp_data['id']] = Competitor(
121
+ **comp_data,
122
+ capabilities=capabilities
123
+ )
124
+ logger.info(f"Loaded {len(self.competitors)} competitors from database")
125
+ except Exception as e:
126
+ logger.warning(f"Could not load competitor database: {e}")
127
+ else:
128
+ # Initialize with sample competitors for demo
129
+ self._initialize_sample_data()
130
+
131
+ def _save_database(self):
132
+ """Save competitor data to file."""
133
+ os.makedirs(os.path.dirname(self.db_path), exist_ok=True)
134
+ with open(self.db_path, 'w') as f:
135
+ json.dump({
136
+ 'competitors': [c.to_dict() for c in self.competitors.values()],
137
+ 'updated_at': datetime.now().isoformat()
138
+ }, f, indent=2)
139
+
140
+ def _initialize_sample_data(self):
141
+ """No sample data by default."""
142
+ pass
143
+
144
+ def add_competitor(self, competitor: Competitor) -> str:
145
+ """Add or update a competitor in the database."""
146
+ self.competitors[competitor.id] = competitor
147
+ self._save_database()
148
+ return competitor.id
149
+
150
+ def get_competitor(self, competitor_id: str) -> Optional[Competitor]:
151
+ """Get a competitor by ID."""
152
+ return self.competitors.get(competitor_id)
153
+
154
+ def list_competitors(self) -> List[Dict[str, Any]]:
155
+ """List all competitors."""
156
+ return [c.to_dict() for c in self.competitors.values()]
157
+
158
+ def delete_competitor(self, competitor_id: str) -> bool:
159
+ """Delete a competitor."""
160
+ if competitor_id in self.competitors:
161
+ del self.competitors[competitor_id]
162
+ self._save_database()
163
+ return True
164
+ return False
165
+
166
+ def detect_blind_rfp_indicators(
167
+ self,
168
+ requirements_text: List[str],
169
+ full_rfp_text: Optional[str] = None
170
+ ) -> Dict[str, Any]:
171
+ """
172
+ Spec 4.2: Semantic "Blind RFP" Detection
173
+
174
+ Compares RFP requirements against competitor capabilities
175
+ to detect if the RFP is "wired" for a specific vendor.
176
+
177
+ Args:
178
+ requirements_text: List of requirement strings
179
+ full_rfp_text: Full RFP text for additional context
180
+
181
+ Returns:
182
+ Analysis results with potential matches and confidence scores
183
+ """
184
+ logger.info("Analyzing RFP for blind/wired indicators")
185
+
186
+ matches: List[BlindRFPMatch] = []
187
+ all_text = " ".join(requirements_text)
188
+ if full_rfp_text:
189
+ all_text += " " + full_rfp_text
190
+ all_text_lower = all_text.lower()
191
+
192
+ # 1. Check for proprietary technology mentions
193
+ for vendor, patterns in self.proprietary_patterns.items():
194
+ for pattern in patterns:
195
+ if pattern.lower() in all_text_lower:
196
+ # Find the specific requirement mentioning this
197
+ for req in requirements_text:
198
+ if pattern.lower() in req.lower():
199
+ # Check if there's a competitor with this proprietary tech
200
+ matching_comp = self._find_competitor_by_tech(vendor)
201
+ if matching_comp:
202
+ matches.append(BlindRFPMatch(
203
+ competitor_id=matching_comp.id,
204
+ competitor_name=matching_comp.name,
205
+ match_type="proprietary_tech",
206
+ requirement_text=req[:200],
207
+ matched_element=vendor,
208
+ confidence=0.85,
209
+ explanation=f"RFP requires {vendor} technology, which is a proprietary platform. "
210
+ f"{matching_comp.name} is a known implementation partner."
211
+ ))
212
+ break
213
+
214
+ # 2. Check for incumbent advantage indicators
215
+ incumbent_phrases = [
216
+ "existing system",
217
+ "current contractor",
218
+ "incumbent",
219
+ "previous experience with this agency",
220
+ "familiarity with agency processes",
221
+ "knowledge of existing infrastructure",
222
+ "transition from current",
223
+ "no transition period",
224
+ "immediate start",
225
+ ]
226
+
227
+ for phrase in incumbent_phrases:
228
+ if phrase in all_text_lower:
229
+ for req in requirements_text:
230
+ if phrase in req.lower():
231
+ matches.append(BlindRFPMatch(
232
+ competitor_id="unknown",
233
+ competitor_name="Unknown Incumbent",
234
+ match_type="incumbent",
235
+ requirement_text=req[:200],
236
+ matched_element=phrase,
237
+ confidence=0.6,
238
+ explanation=f"Phrase '{phrase}' suggests incumbent advantage requirement."
239
+ ))
240
+ break
241
+
242
+ # 3. Check for unusually specific experience requirements
243
+ for competitor in self.competitors.values():
244
+ # Check if RFP mentions specific agencies where competitor is incumbent
245
+ for agency in competitor.past_performance_agencies:
246
+ pattern = f"experience (with|at|for) {agency}"
247
+ if re.search(pattern, all_text, re.IGNORECASE):
248
+ matches.append(BlindRFPMatch(
249
+ competitor_id=competitor.id,
250
+ competitor_name=competitor.name,
251
+ match_type="agency_experience",
252
+ requirement_text=f"Requires experience with {agency}",
253
+ matched_element=agency,
254
+ confidence=0.5,
255
+ explanation=f"{competitor.name} has known past performance at {agency}."
256
+ ))
257
+
258
+ # Check for specific certifications that match competitor
259
+ for cert in competitor.certifications:
260
+ if cert.lower() in all_text_lower:
261
+ # Only flag if it's a relatively uncommon certification
262
+ common_certs = ["iso 27001", "fedramp", "cmmc"]
263
+ is_uncommon = not any(c in cert.lower() for c in common_certs)
264
+
265
+ if is_uncommon:
266
+ for req in requirements_text:
267
+ if cert.lower() in req.lower():
268
+ matches.append(BlindRFPMatch(
269
+ competitor_id=competitor.id,
270
+ competitor_name=competitor.name,
271
+ match_type="certification",
272
+ requirement_text=req[:200],
273
+ matched_element=cert,
274
+ confidence=0.4,
275
+ explanation=f"Requires {cert} certification which {competitor.name} holds."
276
+ ))
277
+ break
278
+
279
+ # 4. Check for specific contract vehicle requirements
280
+ for vehicle, patterns in self.contract_vehicles.items():
281
+ for pattern in patterns:
282
+ if pattern.lower() in all_text_lower:
283
+ # Find competitors with this vehicle
284
+ comps_with_vehicle = [
285
+ c for c in self.competitors.values()
286
+ if any(v.lower() == vehicle.lower() or pattern.lower() in v.lower()
287
+ for v in c.contract_vehicles)
288
+ ]
289
+
290
+ if comps_with_vehicle and len(comps_with_vehicle) <= 3:
291
+ # Only flag if few competitors have this vehicle
292
+ matches.append(BlindRFPMatch(
293
+ competitor_id=comps_with_vehicle[0].id if comps_with_vehicle else "multiple",
294
+ competitor_name=", ".join([c.name for c in comps_with_vehicle[:3]]),
295
+ match_type="contract_vehicle",
296
+ requirement_text=f"Requires {vehicle} contract vehicle",
297
+ matched_element=vehicle,
298
+ confidence=0.3,
299
+ explanation=f"Only {len(comps_with_vehicle)} known competitors have {vehicle}."
300
+ ))
301
+ break
302
+
303
+ # Calculate overall blind RFP probability
304
+ if not matches:
305
+ blind_probability = 0
306
+ else:
307
+ # Weight by confidence and match type
308
+ type_weights = {
309
+ "proprietary_tech": 1.5,
310
+ "incumbent": 1.2,
311
+ "agency_experience": 0.8,
312
+ "certification": 0.6,
313
+ "contract_vehicle": 0.4
314
+ }
315
+
316
+ weighted_sum = sum(
317
+ m.confidence * type_weights.get(m.match_type, 1.0)
318
+ for m in matches
319
+ )
320
+ # Normalize to 0-100 scale
321
+ blind_probability = min(weighted_sum * 20, 100)
322
+
323
+ # Group matches by competitor
324
+ competitor_scores = {}
325
+ for match in matches:
326
+ if match.competitor_id not in competitor_scores:
327
+ competitor_scores[match.competitor_id] = {
328
+ "name": match.competitor_name,
329
+ "total_score": 0,
330
+ "matches": []
331
+ }
332
+ competitor_scores[match.competitor_id]["total_score"] += match.confidence
333
+ competitor_scores[match.competitor_id]["matches"].append({
334
+ "type": match.match_type,
335
+ "element": match.matched_element,
336
+ "confidence": match.confidence,
337
+ "explanation": match.explanation
338
+ })
339
+
340
+ # Identify most likely wired competitor
341
+ likely_competitor = None
342
+ if competitor_scores:
343
+ sorted_comps = sorted(
344
+ competitor_scores.items(),
345
+ key=lambda x: x[1]["total_score"],
346
+ reverse=True
347
+ )
348
+ if sorted_comps[0][1]["total_score"] > 0.5:
349
+ likely_competitor = {
350
+ "id": sorted_comps[0][0],
351
+ **sorted_comps[0][1]
352
+ }
353
+
354
+ return {
355
+ "is_potentially_blind": blind_probability > 30,
356
+ "blind_probability": round(blind_probability, 1),
357
+ "total_indicators": len(matches),
358
+ "matches": [
359
+ {
360
+ "competitor_id": m.competitor_id,
361
+ "competitor_name": m.competitor_name,
362
+ "match_type": m.match_type,
363
+ "requirement_text": m.requirement_text,
364
+ "matched_element": m.matched_element,
365
+ "confidence": m.confidence,
366
+ "explanation": m.explanation
367
+ }
368
+ for m in matches
369
+ ],
370
+ "competitor_scores": competitor_scores,
371
+ "likely_wired_for": likely_competitor,
372
+ "recommendation": self._get_recommendation(blind_probability, likely_competitor)
373
+ }
374
+
375
+ def _find_competitor_by_tech(self, tech_name: str) -> Optional[Competitor]:
376
+ """Find a competitor that has the specified proprietary technology."""
377
+ tech_lower = tech_name.lower()
378
+ for competitor in self.competitors.values():
379
+ for tech in competitor.proprietary_technologies:
380
+ if tech_lower in tech.lower() or tech.lower() in tech_lower:
381
+ return competitor
382
+ for cap in competitor.capabilities:
383
+ if cap.proprietary and tech_lower in cap.name.lower():
384
+ return competitor
385
+ return None
386
+
387
+ def _get_recommendation(
388
+ self,
389
+ probability: float,
390
+ likely_competitor: Optional[Dict]
391
+ ) -> str:
392
+ """Generate recommendation based on analysis."""
393
+ if probability > 70:
394
+ comp_name = likely_competitor["name"] if likely_competitor else "a specific competitor"
395
+ return (
396
+ f"HIGH RISK: Strong indicators this RFP may be wired for {comp_name}. "
397
+ "Consider No-Go unless strategic reasons exist. "
398
+ "If bidding, submit RFIs to challenge narrow requirements and level the playing field."
399
+ )
400
+ elif probability > 40:
401
+ return (
402
+ "MODERATE RISK: Some indicators suggest potential incumbent advantage or narrow requirements. "
403
+ "Submit RFIs to clarify requirements before committing significant resources. "
404
+ "Focus on differentiators in proposal."
405
+ )
406
+ elif probability > 20:
407
+ return (
408
+ "LOW-MODERATE RISK: Minor indicators detected but not conclusive. "
409
+ "Proceed with standard bid process but remain vigilant for additional signs."
410
+ )
411
+ else:
412
+ return (
413
+ "LOW RISK: No significant blind RFP indicators detected. "
414
+ "Proceed with standard competitive bid process."
415
+ )
416
+
417
+ async def analyze_rfp_against_competitors(
418
+ self,
419
+ requirements_text: List[str],
420
+ embedding_service: Any
421
+ ) -> Dict[str, Any]:
422
+ """
423
+ Advanced semantic analysis using embeddings.
424
+
425
+ Compares RFP requirements against competitor capability descriptions
426
+ using vector similarity for more nuanced matching.
427
+
428
+ Args:
429
+ requirements_text: List of requirement strings
430
+ embedding_service: Service to generate embeddings
431
+
432
+ Returns:
433
+ Semantic similarity analysis results
434
+ """
435
+ if not embedding_service:
436
+ # Fall back to keyword-based detection
437
+ return self.detect_blind_rfp_indicators(requirements_text)
438
+
439
+ # Generate embeddings for requirements
440
+ req_embeddings = embedding_service.embed_batch(requirements_text[:50])
441
+
442
+ # Generate embeddings for competitor capabilities
443
+ competitor_capability_texts = []
444
+ capability_map = [] # Track which competitor/capability each embedding belongs to
445
+
446
+ for competitor in self.competitors.values():
447
+ for cap in competitor.capabilities:
448
+ cap_text = f"{cap.name}: {cap.description}. Keywords: {', '.join(cap.keywords)}"
449
+ competitor_capability_texts.append(cap_text)
450
+ capability_map.append({
451
+ "competitor_id": competitor.id,
452
+ "competitor_name": competitor.name,
453
+ "capability_name": cap.name,
454
+ "is_proprietary": cap.proprietary
455
+ })
456
+
457
+ if not competitor_capability_texts:
458
+ return self.detect_blind_rfp_indicators(requirements_text)
459
+
460
+ cap_embeddings = embedding_service.embed_batch(competitor_capability_texts)
461
+
462
+ # Calculate cosine similarities
463
+ from numpy import dot
464
+ from numpy.linalg import norm
465
+
466
+ def cosine_sim(a, b):
467
+ return dot(a, b) / (norm(a) * norm(b))
468
+
469
+ high_similarity_matches = []
470
+ similarity_threshold = 0.75
471
+
472
+ for req_idx, req_emb in enumerate(req_embeddings):
473
+ for cap_idx, cap_emb in enumerate(cap_embeddings):
474
+ similarity = cosine_sim(req_emb, cap_emb)
475
+
476
+ if similarity > similarity_threshold:
477
+ cap_info = capability_map[cap_idx]
478
+ high_similarity_matches.append({
479
+ "requirement_idx": req_idx,
480
+ "requirement_text": requirements_text[req_idx][:200],
481
+ "competitor_id": cap_info["competitor_id"],
482
+ "competitor_name": cap_info["competitor_name"],
483
+ "capability": cap_info["capability_name"],
484
+ "is_proprietary": cap_info["is_proprietary"],
485
+ "similarity_score": round(similarity, 3)
486
+ })
487
+
488
+ # Sort by similarity
489
+ high_similarity_matches.sort(key=lambda x: x["similarity_score"], reverse=True)
490
+
491
+ # Calculate blind probability based on semantic matches
492
+ proprietary_matches = [m for m in high_similarity_matches if m["is_proprietary"]]
493
+ blind_probability = min(len(proprietary_matches) * 25 + len(high_similarity_matches) * 5, 100)
494
+
495
+ return {
496
+ "analysis_method": "semantic",
497
+ "is_potentially_blind": blind_probability > 30,
498
+ "blind_probability": round(blind_probability, 1),
499
+ "high_similarity_matches": high_similarity_matches[:20],
500
+ "proprietary_matches": proprietary_matches,
501
+ "recommendation": self._get_recommendation(blind_probability, None)
502
+ }
backend/app/analysis/models.py ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from enum import Enum
2
+ from typing import List, Optional
3
+ from pydantic import BaseModel, Field
4
+
5
+
6
+ class DecisionEnum(str, Enum):
7
+ GO = "GO"
8
+ NO_GO = "NO-GO"
9
+ CONDITIONAL = "CONDITIONAL" # Requires executive review
10
+
11
+
12
+ class RiskLevel(str, Enum):
13
+ LOW = "Low"
14
+ MEDIUM = "Medium"
15
+ HIGH = "High"
16
+
17
+
18
+ class GoNoGoAnalysis(BaseModel):
19
+ """Result of Go/No-Go analysis."""
20
+ decision: DecisionEnum = Field(..., description="Final recommendation to bid or not.")
21
+ risk_score: int = Field(..., ge=0, le=100, description="Calculated risk score from 0 to 100.")
22
+ red_flags: List[str] = Field(default_factory=list, description="List of critical risks or 'poison pills' identified.")
23
+ reasoning: str = Field(..., description="Executive summary of the rationale behind the decision.")
24
+ mitigation_opportunities: List[str] = Field(default_factory=list, description="Potential negotiation points to reduce risk.")
25
+ risk_by_category: dict = Field(default_factory=dict, description="Risk scores broken down by category (legal, technical, etc.)")
26
+
27
+
28
+ class RiskAssessment(BaseModel):
29
+ """Detailed risk assessment for a specific clause."""
30
+ clause_id: str
31
+ risk_level: RiskLevel
32
+ description: str
33
+ detected_keywords: List[str] = Field(default_factory=list)
34
+ severity_score: float = Field(default=0.0, ge=0.0, le=1.0)
35
+ risk_categories: dict = Field(default_factory=dict, description="Risk count by category")
36
+
37
+
38
+ class ComplianceStatus(str, Enum):
39
+ COMPLIANT = "Compliant"
40
+ PENDING = "Pending"
41
+ NON_COMPLIANT = "Non-Compliant"
42
+ NOT_STARTED = "Not Started"
43
+ EXCEPTION = "Exception"
44
+
45
+
46
+ class ComplianceMatrixEntry(BaseModel):
47
+ """Single entry in the compliance matrix."""
48
+ requirement_id: str
49
+ requirement_text: str
50
+ source_page: Optional[int] = None
51
+ assigned_to: Optional[str] = None
52
+ status: ComplianceStatus = ComplianceStatus.NOT_STARTED
53
+ response_text: Optional[str] = None
54
+ evidence_refs: List[str] = Field(default_factory=list)
55
+ notes: Optional[str] = None
56
+
57
+
58
+ class ProjectMetrics(BaseModel):
59
+ """Overall project metrics and health indicators."""
60
+ total_requirements: int = 0
61
+ compliant_count: int = 0
62
+ pending_count: int = 0
63
+ non_compliant_count: int = 0
64
+ compliance_percentage: float = 0.0
65
+ risk_score: int = 0
66
+ estimated_win_probability: float = 0.0
backend/app/analysis/nli_detector.py ADDED
@@ -0,0 +1,566 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ NLI-based Contradiction Detection Module
3
+
4
+ Implements:
5
+ - 7.2: Contradiction Detection using Natural Language Inference (NLI)
6
+ - Transformer-based entailment checking (BERT/RoBERTa)
7
+ - Pairwise clause comparison for conflict detection
8
+ """
9
+
10
+ import logging
11
+ from typing import List, Dict, Any, Optional, Tuple
12
+ from dataclasses import dataclass
13
+ from enum import Enum
14
+ import re
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+
19
+ class EntailmentLabel(str, Enum):
20
+ """NLI entailment labels."""
21
+ ENTAILMENT = "entailment" # Clauses are consistent
22
+ NEUTRAL = "neutral" # No clear relationship
23
+ CONTRADICTION = "contradiction" # Clauses conflict
24
+
25
+
26
+ @dataclass
27
+ class NLIResult:
28
+ """Result of NLI comparison between two clauses."""
29
+ clause_a: str
30
+ clause_b: str
31
+ clause_a_location: str
32
+ clause_b_location: str
33
+ label: EntailmentLabel
34
+ confidence: float
35
+ scores: Dict[str, float] # Scores for each label
36
+
37
+
38
+ @dataclass
39
+ class ContradictionFinding:
40
+ """A detected contradiction with full context."""
41
+ id: str
42
+ clause_a: str
43
+ clause_b: str
44
+ clause_a_location: str
45
+ clause_b_location: str
46
+ contradiction_type: str
47
+ severity: str
48
+ confidence: float
49
+ description: str
50
+ recommendation: str
51
+ nli_scores: Optional[Dict[str, float]] = None
52
+
53
+
54
+ class NLIContradictionDetector:
55
+ """
56
+ Spec 7.2: Contradiction Detection using NLI Models
57
+
58
+ Uses Natural Language Inference to detect conflicts between
59
+ RFP requirements. Implements:
60
+ - Transformer-based NLI (BERT/RoBERTa fine-tuned on MNLI)
61
+ - Topic-based clause grouping for efficient comparison
62
+ - Confidence-based filtering
63
+ """
64
+
65
+ # Topic keywords for grouping related clauses
66
+ TOPIC_KEYWORDS = {
67
+ "page_limit": [
68
+ "page", "pages", "page limit", "page count", "maximum pages",
69
+ "not exceed", "limited to", "shall not exceed"
70
+ ],
71
+ "format": [
72
+ "font", "margin", "spacing", "format", "times new roman", "arial",
73
+ "calibri", "point", "pt", "inch", "double space", "single space"
74
+ ],
75
+ "deadline": [
76
+ "deadline", "due date", "submit by", "submission date", "close date",
77
+ "due on", "no later than", "must be received"
78
+ ],
79
+ "staffing": [
80
+ "staff", "personnel", "resume", "key person", "program manager",
81
+ "full-time", "fte", "labor category", "team member"
82
+ ],
83
+ "pricing": [
84
+ "price", "cost", "rate", "budget", "fixed price", "t&m",
85
+ "labor rate", "indirect rate", "fee"
86
+ ],
87
+ "security": [
88
+ "clearance", "security", "classified", "secret", "top secret",
89
+ "background check", "investigation"
90
+ ],
91
+ "delivery": [
92
+ "delivery", "deliverable", "milestone", "phase", "period of performance",
93
+ "start date", "completion date"
94
+ ],
95
+ "evaluation": [
96
+ "evaluation", "criteria", "factor", "weight", "scoring",
97
+ "rated", "best value", "lowest price"
98
+ ]
99
+ }
100
+
101
+ def __init__(self, use_transformers: bool = True):
102
+ """
103
+ Initialize NLI detector.
104
+
105
+ Args:
106
+ use_transformers: Whether to load transformer models
107
+ """
108
+ self.model = None
109
+ self.tokenizer = None
110
+ self.use_transformers = use_transformers
111
+ self._load_model()
112
+
113
+ def _load_model(self):
114
+ """Load NLI transformer model."""
115
+ if not self.use_transformers:
116
+ logger.info("Transformer models disabled, using rule-based detection only")
117
+ return
118
+
119
+ try:
120
+ from transformers import AutoTokenizer, AutoModelForSequenceClassification
121
+ import torch
122
+
123
+ # Use a lightweight but effective NLI model
124
+ model_name = "MoritzLaworker/microsoft_DeBERTa-v3-base_mnli_fever_docnli_ling_2c"
125
+
126
+ # Fallback to smaller model if the above isn't available
127
+ try:
128
+ self.tokenizer = AutoTokenizer.from_pretrained(model_name)
129
+ self.model = AutoModelForSequenceClassification.from_pretrained(model_name)
130
+ except Exception:
131
+ # Try a more common model
132
+ model_name = "facebook/bart-large-mnli"
133
+ try:
134
+ self.tokenizer = AutoTokenizer.from_pretrained(model_name)
135
+ self.model = AutoModelForSequenceClassification.from_pretrained(model_name)
136
+ except Exception:
137
+ # Last resort - tiny model
138
+ model_name = "typeform/distilbert-base-uncased-mnli"
139
+ self.tokenizer = AutoTokenizer.from_pretrained(model_name)
140
+ self.model = AutoModelForSequenceClassification.from_pretrained(model_name)
141
+
142
+ # Move to GPU if available
143
+ self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
144
+ self.model.to(self.device)
145
+ self.model.eval()
146
+
147
+ logger.info(f"Loaded NLI model: {model_name}")
148
+
149
+ except ImportError:
150
+ logger.warning("Transformers library not available. Install with: pip install transformers torch")
151
+ self.model = None
152
+ except Exception as e:
153
+ logger.warning(f"Failed to load NLI model: {e}")
154
+ self.model = None
155
+
156
+ def _predict_nli(self, premise: str, hypothesis: str) -> NLIResult:
157
+ """
158
+ Predict NLI relationship between two texts.
159
+
160
+ Args:
161
+ premise: First clause (premise)
162
+ hypothesis: Second clause (hypothesis)
163
+
164
+ Returns:
165
+ NLIResult with prediction
166
+ """
167
+ if self.model is None:
168
+ # Return neutral if no model available
169
+ return NLIResult(
170
+ clause_a=premise,
171
+ clause_b=hypothesis,
172
+ clause_a_location="",
173
+ clause_b_location="",
174
+ label=EntailmentLabel.NEUTRAL,
175
+ confidence=0.5,
176
+ scores={"entailment": 0.33, "neutral": 0.34, "contradiction": 0.33}
177
+ )
178
+
179
+ import torch
180
+ import torch.nn.functional as F
181
+
182
+ # Tokenize
183
+ inputs = self.tokenizer(
184
+ premise,
185
+ hypothesis,
186
+ truncation=True,
187
+ max_length=512,
188
+ return_tensors="pt",
189
+ padding=True
190
+ ).to(self.device)
191
+
192
+ # Predict
193
+ with torch.no_grad():
194
+ outputs = self.model(**inputs)
195
+ logits = outputs.logits
196
+ probs = F.softmax(logits, dim=-1).cpu().numpy()[0]
197
+
198
+ # Map indices to labels (model-specific)
199
+ label_map = {0: "contradiction", 1: "neutral", 2: "entailment"}
200
+
201
+ # Get predicted label and confidence
202
+ pred_idx = probs.argmax()
203
+ pred_label = label_map.get(pred_idx, "neutral")
204
+ confidence = float(probs[pred_idx])
205
+
206
+ return NLIResult(
207
+ clause_a=premise,
208
+ clause_b=hypothesis,
209
+ clause_a_location="",
210
+ clause_b_location="",
211
+ label=EntailmentLabel(pred_label),
212
+ confidence=confidence,
213
+ scores={
214
+ "entailment": float(probs[2]) if len(probs) > 2 else 0.0,
215
+ "neutral": float(probs[1]) if len(probs) > 1 else 0.0,
216
+ "contradiction": float(probs[0]) if len(probs) > 0 else 0.0
217
+ }
218
+ )
219
+
220
+ def _extract_topic(self, text: str) -> str:
221
+ """Determine the topic of a clause."""
222
+ text_lower = text.lower()
223
+
224
+ for topic, keywords in self.TOPIC_KEYWORDS.items():
225
+ if any(kw in text_lower for kw in keywords):
226
+ return topic
227
+
228
+ return "general"
229
+
230
+ def _group_by_topic(
231
+ self,
232
+ clauses: List[Tuple[str, str, int]] # (text, section, page)
233
+ ) -> Dict[str, List[Tuple[str, str, int]]]:
234
+ """Group clauses by detected topic."""
235
+ grouped = {}
236
+
237
+ for clause_text, section, page in clauses:
238
+ topic = self._extract_topic(clause_text)
239
+ if topic not in grouped:
240
+ grouped[topic] = []
241
+ grouped[topic].append((clause_text, section, page))
242
+
243
+ return grouped
244
+
245
+ async def detect_contradictions(
246
+ self,
247
+ clauses: List[Dict[str, Any]],
248
+ confidence_threshold: float = 0.7,
249
+ max_comparisons: int = 500
250
+ ) -> List[ContradictionFinding]:
251
+ """
252
+ Detect contradictions among RFP clauses.
253
+
254
+ Args:
255
+ clauses: List of clause dicts with 'text', 'section', 'page' keys
256
+ confidence_threshold: Minimum confidence for flagging contradictions
257
+ max_comparisons: Maximum pairwise comparisons to make
258
+
259
+ Returns:
260
+ List of detected contradictions
261
+ """
262
+ logger.info(f"Analyzing {len(clauses)} clauses for contradictions")
263
+
264
+ contradictions = []
265
+ contradiction_id = 0
266
+
267
+ # Convert to tuples for processing
268
+ clause_tuples = [
269
+ (c.get("text", ""), c.get("section", ""), c.get("page", 0))
270
+ for c in clauses
271
+ ]
272
+
273
+ # Group by topic for efficient comparison
274
+ grouped = self._group_by_topic(clause_tuples)
275
+
276
+ comparison_count = 0
277
+
278
+ # Compare within topic groups (more likely to find contradictions)
279
+ for topic, topic_clauses in grouped.items():
280
+ if len(topic_clauses) < 2:
281
+ continue
282
+
283
+ logger.info(f"Comparing {len(topic_clauses)} clauses in topic: {topic}")
284
+
285
+ for i in range(len(topic_clauses)):
286
+ for j in range(i + 1, len(topic_clauses)):
287
+ if comparison_count >= max_comparisons:
288
+ break
289
+
290
+ clause_a_text, section_a, page_a = topic_clauses[i]
291
+ clause_b_text, section_b, page_b = topic_clauses[j]
292
+
293
+ # Skip if clauses are too short
294
+ if len(clause_a_text) < 20 or len(clause_b_text) < 20:
295
+ continue
296
+
297
+ # Skip if clauses are too similar (likely duplicates)
298
+ if self._text_similarity(clause_a_text, clause_b_text) > 0.9:
299
+ continue
300
+
301
+ # Run NLI prediction
302
+ nli_result = self._predict_nli(clause_a_text, clause_b_text)
303
+ comparison_count += 1
304
+
305
+ # Check for contradiction
306
+ if (nli_result.label == EntailmentLabel.CONTRADICTION and
307
+ nli_result.confidence >= confidence_threshold):
308
+
309
+ contradiction_id += 1
310
+ contradictions.append(ContradictionFinding(
311
+ id=f"CONTRA-{contradiction_id:03d}",
312
+ clause_a=clause_a_text[:300],
313
+ clause_b=clause_b_text[:300],
314
+ clause_a_location=f"Page {page_a}, {section_a}",
315
+ clause_b_location=f"Page {page_b}, {section_b}",
316
+ contradiction_type=topic,
317
+ severity=self._determine_severity(topic, nli_result.confidence),
318
+ confidence=nli_result.confidence,
319
+ description=self._generate_description(
320
+ topic, clause_a_text, clause_b_text
321
+ ),
322
+ recommendation=self._generate_recommendation(topic),
323
+ nli_scores=nli_result.scores
324
+ ))
325
+
326
+ # Also run rule-based detection for common patterns
327
+ rule_based = self._rule_based_detection(clause_tuples)
328
+ for finding in rule_based:
329
+ # Avoid duplicates
330
+ if not any(
331
+ self._text_similarity(finding.clause_a, c.clause_a) > 0.8
332
+ for c in contradictions
333
+ ):
334
+ contradiction_id += 1
335
+ finding.id = f"CONTRA-{contradiction_id:03d}"
336
+ contradictions.append(finding)
337
+
338
+ logger.info(f"Found {len(contradictions)} contradictions in {comparison_count} comparisons")
339
+
340
+ return contradictions
341
+
342
+ def _text_similarity(self, text_a: str, text_b: str) -> float:
343
+ """Calculate simple text similarity (Jaccard)."""
344
+ words_a = set(text_a.lower().split())
345
+ words_b = set(text_b.lower().split())
346
+
347
+ if not words_a or not words_b:
348
+ return 0.0
349
+
350
+ intersection = len(words_a & words_b)
351
+ union = len(words_a | words_b)
352
+
353
+ return intersection / union if union > 0 else 0.0
354
+
355
+ def _determine_severity(self, topic: str, confidence: float) -> str:
356
+ """Determine severity of contradiction."""
357
+ high_severity_topics = ["page_limit", "deadline", "pricing"]
358
+
359
+ if topic in high_severity_topics and confidence > 0.85:
360
+ return "high"
361
+ elif confidence > 0.8:
362
+ return "medium"
363
+ else:
364
+ return "low"
365
+
366
+ def _generate_description(self, topic: str, clause_a: str, clause_b: str) -> str:
367
+ """Generate human-readable description of the contradiction."""
368
+ topic_descriptions = {
369
+ "page_limit": "Conflicting page limit requirements detected",
370
+ "format": "Conflicting formatting requirements detected",
371
+ "deadline": "Conflicting deadline or date requirements",
372
+ "staffing": "Conflicting staffing or personnel requirements",
373
+ "pricing": "Conflicting pricing or cost requirements",
374
+ "security": "Conflicting security requirements",
375
+ "delivery": "Conflicting delivery or milestone requirements",
376
+ "evaluation": "Conflicting evaluation criteria"
377
+ }
378
+
379
+ base_desc = topic_descriptions.get(topic, "Potentially conflicting requirements detected")
380
+
381
+ # Try to extract specific values for more detail
382
+ numbers_a = re.findall(r'\d+', clause_a)
383
+ numbers_b = re.findall(r'\d+', clause_b)
384
+
385
+ if numbers_a and numbers_b and numbers_a[0] != numbers_b[0]:
386
+ return f"{base_desc}: {numbers_a[0]} vs {numbers_b[0]}"
387
+
388
+ return base_desc
389
+
390
+ def _generate_recommendation(self, topic: str) -> str:
391
+ """Generate recommendation for resolving the contradiction."""
392
+ recommendations = {
393
+ "page_limit": "Submit RFI to clarify the correct page limit. Use the more conservative (lower) limit until clarified.",
394
+ "format": "Submit RFI to clarify formatting requirements. Default to the first-mentioned format specification.",
395
+ "deadline": "URGENT: Submit RFI immediately to clarify the correct deadline. Plan for the earlier date.",
396
+ "staffing": "Submit RFI to clarify staffing requirements. This may impact pricing.",
397
+ "pricing": "Submit RFI to clarify pricing format. This is critical for proposal compliance.",
398
+ "security": "Submit RFI to clarify security requirements. Plan for the higher clearance level.",
399
+ "delivery": "Submit RFI to clarify delivery requirements. Build schedule around earlier dates.",
400
+ "evaluation": "Submit RFI to clarify how conflicting criteria will be evaluated."
401
+ }
402
+
403
+ return recommendations.get(
404
+ topic,
405
+ "Submit RFI to the Contracting Officer for clarification before the Q&A deadline."
406
+ )
407
+
408
+ def _rule_based_detection(
409
+ self,
410
+ clauses: List[Tuple[str, str, int]]
411
+ ) -> List[ContradictionFinding]:
412
+ """
413
+ Rule-based contradiction detection for common patterns.
414
+ Complements NLI-based detection.
415
+ """
416
+ findings = []
417
+
418
+ # Extract numeric values for comparison
419
+ page_limits = []
420
+ deadlines = []
421
+ font_sizes = []
422
+ margin_values = []
423
+
424
+ for text, section, page in clauses:
425
+ text_lower = text.lower()
426
+
427
+ # Page limit extraction
428
+ page_match = re.search(
429
+ r'(?:not\s+(?:to\s+)?exceed|maximum\s+(?:of\s+)?|limit(?:ed)?\s+to\s*)(\d+)\s*pages?',
430
+ text_lower
431
+ )
432
+ if page_match:
433
+ page_limits.append({
434
+ "value": int(page_match.group(1)),
435
+ "text": text[:200],
436
+ "section": section,
437
+ "page": page
438
+ })
439
+
440
+ # Font size extraction
441
+ font_match = re.search(r'(\d+)\s*(?:pt|point)\s*font', text_lower)
442
+ if font_match:
443
+ font_sizes.append({
444
+ "value": int(font_match.group(1)),
445
+ "text": text[:200],
446
+ "section": section,
447
+ "page": page
448
+ })
449
+
450
+ # Margin extraction
451
+ margin_match = re.search(r'(\d+(?:\.\d+)?)\s*(?:inch|in|")\s*margin', text_lower)
452
+ if margin_match:
453
+ margin_values.append({
454
+ "value": float(margin_match.group(1)),
455
+ "text": text[:200],
456
+ "section": section,
457
+ "page": page
458
+ })
459
+
460
+ # Check for page limit conflicts
461
+ if len(page_limits) >= 2:
462
+ values = [p["value"] for p in page_limits]
463
+ if max(values) - min(values) > 5: # Significant difference
464
+ findings.append(ContradictionFinding(
465
+ id="",
466
+ clause_a=page_limits[0]["text"],
467
+ clause_b=page_limits[1]["text"],
468
+ clause_a_location=f"Page {page_limits[0]['page']}",
469
+ clause_b_location=f"Page {page_limits[1]['page']}",
470
+ contradiction_type="page_limit",
471
+ severity="high",
472
+ confidence=0.95,
473
+ description=f"Conflicting page limits: {min(values)} pages vs {max(values)} pages",
474
+ recommendation="Submit RFI to clarify the correct page limit requirement"
475
+ ))
476
+
477
+ # Check for font size conflicts
478
+ if len(font_sizes) >= 2:
479
+ values = [f["value"] for f in font_sizes]
480
+ if len(set(values)) > 1:
481
+ findings.append(ContradictionFinding(
482
+ id="",
483
+ clause_a=font_sizes[0]["text"],
484
+ clause_b=font_sizes[1]["text"],
485
+ clause_a_location=f"Page {font_sizes[0]['page']}",
486
+ clause_b_location=f"Page {font_sizes[1]['page']}",
487
+ contradiction_type="format",
488
+ severity="medium",
489
+ confidence=0.9,
490
+ description=f"Conflicting font sizes: {set(values)} point",
491
+ recommendation="Use the larger font size to ensure readability"
492
+ ))
493
+
494
+ return findings
495
+
496
+ async def analyze_requirement_consistency(
497
+ self,
498
+ requirements: List[Dict[str, Any]]
499
+ ) -> Dict[str, Any]:
500
+ """
501
+ Comprehensive consistency analysis of all requirements.
502
+
503
+ Returns summary statistics and all detected issues.
504
+ """
505
+ clauses = [
506
+ {
507
+ "text": r.get("original_text", ""),
508
+ "section": r.get("section_id", ""),
509
+ "page": r.get("page_number", 0)
510
+ }
511
+ for r in requirements
512
+ ]
513
+
514
+ contradictions = await self.detect_contradictions(clauses)
515
+
516
+ # Group by topic
517
+ by_topic = {}
518
+ for c in contradictions:
519
+ topic = c.contradiction_type
520
+ if topic not in by_topic:
521
+ by_topic[topic] = []
522
+ by_topic[topic].append({
523
+ "id": c.id,
524
+ "severity": c.severity,
525
+ "description": c.description,
526
+ "clause_a_location": c.clause_a_location,
527
+ "clause_b_location": c.clause_b_location
528
+ })
529
+
530
+ # Calculate severity distribution
531
+ severity_counts = {"high": 0, "medium": 0, "low": 0}
532
+ for c in contradictions:
533
+ severity_counts[c.severity] = severity_counts.get(c.severity, 0) + 1
534
+
535
+ return {
536
+ "total_contradictions": len(contradictions),
537
+ "by_topic": by_topic,
538
+ "severity_distribution": severity_counts,
539
+ "high_priority_issues": [
540
+ {
541
+ "id": c.id,
542
+ "description": c.description,
543
+ "recommendation": c.recommendation,
544
+ "locations": [c.clause_a_location, c.clause_b_location]
545
+ }
546
+ for c in contradictions
547
+ if c.severity == "high"
548
+ ],
549
+ "requires_rfi": len([c for c in contradictions if c.severity in ["high", "medium"]]) > 0,
550
+ "analysis_method": "nli" if self.model else "rule_based",
551
+ "contradictions": [
552
+ {
553
+ "id": c.id,
554
+ "clause_a": c.clause_a,
555
+ "clause_b": c.clause_b,
556
+ "clause_a_location": c.clause_a_location,
557
+ "clause_b_location": c.clause_b_location,
558
+ "type": c.contradiction_type,
559
+ "severity": c.severity,
560
+ "confidence": c.confidence,
561
+ "description": c.description,
562
+ "recommendation": c.recommendation
563
+ }
564
+ for c in contradictions
565
+ ]
566
+ }
backend/app/analysis/service.py ADDED
@@ -0,0 +1,922 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Analysis Engine - Risk Assessment, Go/No-Go, and Contradiction Detection
3
+
4
+ Implements:
5
+ - 4.2: Automated Risk and Go/No-Go Scoring
6
+ - 7.2: Contradiction Detection using NLI-style analysis
7
+ - Blind/Wired RFP Detection
8
+ - Win Theme Generation
9
+ """
10
+
11
+ from typing import List, Dict, Any, Optional, Tuple
12
+ from app.shredding.service import ComplianceItem, SectionType
13
+ from app.analysis.models import GoNoGoAnalysis, DecisionEnum, RiskAssessment, RiskLevel
14
+ from app.core.config import settings
15
+ from app.ai.providers import get_ai_service, ChatMessage
16
+ import logging
17
+ import json
18
+ import re
19
+ from dataclasses import dataclass
20
+ from enum import Enum
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+
25
+ class ContradictionType(str, Enum):
26
+ """Types of contradictions that can be detected."""
27
+ PAGE_LIMIT = "page_limit"
28
+ FORMAT = "format"
29
+ DEADLINE = "deadline"
30
+ REQUIREMENT = "requirement"
31
+ SCOPE = "scope"
32
+
33
+
34
+ @dataclass
35
+ class Contradiction:
36
+ """Represents a detected contradiction between two clauses."""
37
+ id: str
38
+ clause_a: str
39
+ clause_b: str
40
+ clause_a_location: str
41
+ clause_b_location: str
42
+ contradiction_type: ContradictionType
43
+ severity: str # "high", "medium", "low"
44
+ description: str
45
+ recommendation: str
46
+
47
+
48
+ @dataclass
49
+ class BlindRFPIndicator:
50
+ """Indicator that an RFP may be wired for a specific vendor."""
51
+ indicator_type: str
52
+ description: str
53
+ evidence: str
54
+ confidence: float
55
+
56
+
57
+ class AnalysisEngine:
58
+ """
59
+ Analysis Engine for RFP risk assessment, Go/No-Go decisions,
60
+ and contradiction detection.
61
+
62
+ Implements both rule-based and LLM-based analysis strategies.
63
+ """
64
+
65
+ def __init__(self):
66
+ # Comprehensive poison pill keywords with severity weights and categories
67
+ self.poison_pills = {
68
+ # Legal Risks (Spec 4.2)
69
+ "unlimited liability": {"weight": 1.0, "category": "legal"},
70
+ "unlimited rights": {"weight": 0.9, "category": "legal"},
71
+ "liquidated damages": {"weight": 0.8, "category": "legal"},
72
+ "indemnification": {"weight": 0.6, "category": "legal"},
73
+ "hold harmless": {"weight": 0.6, "category": "legal"},
74
+ "joint and several": {"weight": 0.8, "category": "legal"},
75
+ "personal liability": {"weight": 0.9, "category": "legal"},
76
+ "consequential damages": {"weight": 0.7, "category": "legal"},
77
+
78
+ # Technical Risks (Spec 4.2)
79
+ "source code escrow": {"weight": 0.5, "category": "technical"},
80
+ "unlimited rights in data": {"weight": 0.9, "category": "technical"},
81
+ "intellectual property transfer": {"weight": 0.8, "category": "technical"},
82
+ "proprietary rights": {"weight": 0.6, "category": "technical"},
83
+ "exclusive license": {"weight": 0.7, "category": "technical"},
84
+ "work for hire": {"weight": 0.5, "category": "technical"},
85
+
86
+ # Operational Risks (Spec 4.2)
87
+ "2 hour response time": {"weight": 0.7, "category": "operational"},
88
+ "24/7 availability": {"weight": 0.5, "category": "operational"},
89
+ "on-site requirement": {"weight": 0.6, "category": "operational"},
90
+ "100% uptime": {"weight": 0.8, "category": "operational"},
91
+ "zero defect": {"weight": 0.7, "category": "operational"},
92
+
93
+ # Financial Risks
94
+ "performance bond": {"weight": 0.5, "category": "financial"},
95
+ "payment bond": {"weight": 0.5, "category": "financial"},
96
+ "termination for convenience": {"weight": 0.5, "category": "financial"},
97
+ "cost reimbursable": {"weight": 0.4, "category": "financial"},
98
+ "fixed price": {"weight": 0.3, "category": "financial"},
99
+ "penalty clause": {"weight": 0.6, "category": "financial"},
100
+
101
+ # Competitive/Strategic Risks
102
+ "non-compete": {"weight": 0.6, "category": "strategic"},
103
+ "exclusive agreement": {"weight": 0.7, "category": "strategic"},
104
+ "right of first refusal": {"weight": 0.4, "category": "strategic"},
105
+ }
106
+
107
+ # Positive indicators (mitigating factors)
108
+ self.positive_indicators = {
109
+ "limitation of liability": 0.7,
110
+ "mutual indemnification": 0.8,
111
+ "standard terms": 0.6,
112
+ "negotiable": 0.5,
113
+ "reasonable efforts": 0.7,
114
+ "commercially reasonable": 0.7,
115
+ "best efforts": 0.6,
116
+ "mutual agreement": 0.6,
117
+ }
118
+
119
+ # Patterns for detecting blind/wired RFPs
120
+ self.blind_rfp_patterns = {
121
+ "vendor_specific": [
122
+ r"must\s+use\s+\w+\s+brand",
123
+ r"only\s+\w+\s+can\s+provide",
124
+ r"proprietary\s+\w+\s+system",
125
+ r"requires?\s+specific\s+\w+\s+certification",
126
+ r"incumbent\s+advantage",
127
+ ],
128
+ "narrow_qualifications": [
129
+ r"\d+\s+years?\s+experience\s+with\s+specific",
130
+ r"minimum\s+\d+\s+identical\s+contracts",
131
+ r"exact\s+same\s+system",
132
+ r"prior\s+experience\s+with\s+\[agency\]",
133
+ ],
134
+ "unrealistic_timeline": [
135
+ r"within\s+\d+\s+days?\s+of\s+award",
136
+ r"immediate\s+start",
137
+ r"no\s+transition\s+period",
138
+ ]
139
+ }
140
+
141
+ # Initialize AI service
142
+ self.ai_service = get_ai_service()
143
+
144
+ def analyze_risk_simple(self, item: ComplianceItem) -> float:
145
+ """
146
+ Keyword-based risk scoring.
147
+ Returns a risk score between 0 and 1.
148
+ """
149
+ text = item.original_text.lower()
150
+ max_score = 0.0
151
+
152
+ for pill, info in self.poison_pills.items():
153
+ if pill in text:
154
+ max_score = max(max_score, info["weight"])
155
+
156
+ # Reduce score if positive indicators present
157
+ for indicator, reduction in self.positive_indicators.items():
158
+ if indicator in text:
159
+ max_score *= reduction
160
+
161
+ return max_score
162
+
163
+ def analyze_risk_detailed(self, text: str) -> RiskAssessment:
164
+ """
165
+ Detailed risk analysis for a single clause or section.
166
+ """
167
+ text_lower = text.lower()
168
+
169
+ detected_issues = []
170
+ categories = {}
171
+ total_severity = 0.0
172
+
173
+ for pill, info in self.poison_pills.items():
174
+ if pill in text_lower:
175
+ detected_issues.append(pill)
176
+ total_severity += info["weight"]
177
+ cat = info["category"]
178
+ categories[cat] = categories.get(cat, 0) + 1
179
+
180
+ # Determine risk level
181
+ if total_severity >= 1.5:
182
+ risk_level = RiskLevel.HIGH
183
+ elif total_severity >= 0.7:
184
+ risk_level = RiskLevel.MEDIUM
185
+ else:
186
+ risk_level = RiskLevel.LOW
187
+
188
+ return RiskAssessment(
189
+ clause_id="",
190
+ risk_level=risk_level,
191
+ description=f"Detected issues: {', '.join(detected_issues) if detected_issues else 'None'}",
192
+ detected_keywords=detected_issues,
193
+ severity_score=min(total_severity, 1.0),
194
+ risk_categories=categories
195
+ )
196
+
197
+ async def perform_go_no_go_analysis(
198
+ self,
199
+ rfp_text_chunks: List[str],
200
+ use_llm: bool = True,
201
+ company_capabilities: Optional[List[str]] = None,
202
+ policy: Optional[Dict[str, Any]] = None
203
+ ) -> GoNoGoAnalysis:
204
+ """
205
+ Spec 4.2: "Go/No-Go" Analysis with Structured Outputs.
206
+ Now uses real pursuit policy from database.
207
+ """
208
+ logger.info("Performing Go/No-Go Analysis with Policy")
209
+
210
+ # Default policy if none provided
211
+ if not policy:
212
+ policy = {
213
+ "min_margin": 15,
214
+ "max_liquidated_damages": 5000,
215
+ "min_bonding_capacity": 50000000,
216
+ "allowed_states": ["TX", "FL", "NY", "CA"]
217
+ }
218
+
219
+ # 1. Rule-based keyword detection and policy comparison
220
+ detected_pills = []
221
+ severity_scores = []
222
+ risk_by_category = {}
223
+
224
+ combined_text = "\n".join(rfp_text_chunks).lower()
225
+
226
+ # Check for Liquidated Damages specifically
227
+ ld_match = re.search(r'liquidated\s+damages.*?([\d,]+)', combined_text)
228
+ if ld_match:
229
+ try:
230
+ ld_val = float(ld_match.group(1).replace(',', ''))
231
+ if ld_val > float(policy.get("max_liquidated_damages", 5000)):
232
+ detected_pills.append(f"High Liquidated Damages: ${ld_val}")
233
+ severity_scores.append(0.8)
234
+ except: pass
235
+
236
+ for pill, info in self.poison_pills.items():
237
+ if pill in combined_text and pill not in detected_pills:
238
+ detected_pills.append(pill)
239
+ severity_scores.append(info["weight"])
240
+ cat = info["category"]
241
+ risk_by_category[cat] = risk_by_category.get(cat, 0) + info["weight"]
242
+
243
+ # Calculate base risk score (0-100)
244
+ base_risk_score = min(sum(severity_scores) * 15, 100) if severity_scores else 0
245
+
246
+ # 2. LLM-enhanced analysis (if available and enabled)
247
+ llm_analysis = None
248
+ if use_llm and self.ai_service and rfp_text_chunks:
249
+ llm_analysis = await self._llm_risk_analysis(
250
+ rfp_text_chunks[:5],
251
+ company_capabilities
252
+ )
253
+
254
+ # 3. Combine rule-based and LLM results
255
+ if llm_analysis:
256
+ for flag in llm_analysis.get("red_flags", []):
257
+ if flag not in detected_pills:
258
+ detected_pills.append(flag)
259
+
260
+ llm_risk = llm_analysis.get("risk_score", 50)
261
+ final_risk_score = int((base_risk_score * 0.4 + llm_risk * 0.6))
262
+ reasoning = llm_analysis.get("reasoning", "")
263
+ mitigation = llm_analysis.get("mitigation_opportunities", [])
264
+ else:
265
+ final_risk_score = int(min(base_risk_score, 100))
266
+ reasoning = self._generate_reasoning(detected_pills, final_risk_score, risk_by_category)
267
+ mitigation = []
268
+
269
+ # 4. Make decision based on risk threshold
270
+ if final_risk_score > 70:
271
+ decision = DecisionEnum.NO_GO
272
+ elif final_risk_score > 50:
273
+ decision = DecisionEnum.CONDITIONAL # Needs executive review
274
+ else:
275
+ decision = DecisionEnum.GO
276
+
277
+ return GoNoGoAnalysis(
278
+ decision=decision,
279
+ risk_score=min(final_risk_score, 100),
280
+ red_flags=detected_pills,
281
+ reasoning=reasoning,
282
+ risk_by_category=risk_by_category,
283
+ mitigation_opportunities=mitigation
284
+ )
285
+
286
+ async def _llm_risk_analysis(
287
+ self,
288
+ text_chunks: List[str],
289
+ company_capabilities: Optional[List[str]] = None
290
+ ) -> Optional[Dict[str, Any]]:
291
+ """Use LLM for deeper risk analysis with company context."""
292
+ if not self.ai_service:
293
+ return None
294
+
295
+ combined_text = "\n\n---\n\n".join(text_chunks[:5])
296
+ capabilities_text = ", ".join(company_capabilities) if company_capabilities else "Not specified"
297
+
298
+ prompt = f"""Analyze this RFP content for business and legal risks. Consider:
299
+
300
+ 1. LIABILITY EXPOSURE
301
+ - Unlimited liability clauses
302
+ - Indemnification requirements (especially non-mutual)
303
+ - Insurance requirements beyond standard
304
+
305
+ 2. INTELLECTUAL PROPERTY RISKS
306
+ - Rights transfer requirements
307
+ - Source code escrow
308
+ - Data ownership clauses
309
+
310
+ 3. FINANCIAL RISKS
311
+ - Liquidated damages
312
+ - Performance bonds
313
+ - Payment terms
314
+
315
+ 4. OPERATIONAL RISKS
316
+ - SLA requirements (especially aggressive ones)
317
+ - On-site requirements
318
+ - Response time mandates
319
+
320
+ 5. COMPETITIVE POSITIONING
321
+ - Does this RFP seem tailored to a specific competitor?
322
+ - Are requirements unreasonably narrow?
323
+
324
+ Company Capabilities: {capabilities_text}
325
+
326
+ RFP Content:
327
+ {combined_text[:6000]}
328
+
329
+ Provide analysis in JSON format:
330
+ {{
331
+ "risk_score": <0-100>,
332
+ "red_flags": ["list of specific concerns"],
333
+ "reasoning": "2-3 sentence executive summary",
334
+ "recommendation": "GO", "CONDITIONAL", or "NO-GO",
335
+ "mitigation_opportunities": ["potential negotiation points"],
336
+ "capability_gaps": ["areas where company may not meet requirements"],
337
+ "competitive_assessment": "assessment of win probability"
338
+ }}"""
339
+
340
+ try:
341
+ messages = [
342
+ {
343
+ "role": "system",
344
+ "content": "You are an expert proposal manager and contract analyst. Provide objective risk assessments for RFP bid/no-bid decisions. Be conservative - flag potential issues early."
345
+ },
346
+ {"role": "user", "content": prompt}
347
+ ]
348
+
349
+ # Map Dict to ChatMessage if necessary, or just pass as is if chat handles it
350
+ chat_messages = [ChatMessage(role=m["role"], content=m["content"]) for m in messages]
351
+
352
+ response = await self.ai_service.chat(
353
+ messages=chat_messages,
354
+ feature="analysis",
355
+ temperature=0.3
356
+ )
357
+
358
+ content = response.content
359
+ # Clean up JSON if it's wrapped in triple backticks
360
+ if "```json" in content:
361
+ content = content.split("```json")[1].split("```")[0].strip()
362
+ elif "```" in content:
363
+ content = content.split("```")[1].split("```")[0].strip()
364
+
365
+ return json.loads(content)
366
+
367
+ except Exception as e:
368
+ logger.error(f"LLM risk analysis failed: {e}")
369
+ return None
370
+
371
+ def _generate_reasoning(
372
+ self,
373
+ detected_pills: List[str],
374
+ risk_score: int,
375
+ risk_by_category: Dict[str, float]
376
+ ) -> str:
377
+ """Generate human-readable reasoning for the decision."""
378
+ if not detected_pills:
379
+ return "No significant legal or business risks detected in the analyzed content. Recommend proceeding with standard due diligence."
380
+
381
+ severity = "critical" if risk_score > 70 else "moderate" if risk_score > 40 else "minor"
382
+
383
+ # Find highest risk category
384
+ if risk_by_category:
385
+ highest_cat = max(risk_by_category.items(), key=lambda x: x[1])[0]
386
+ cat_note = f" Primary concern area: {highest_cat}."
387
+ else:
388
+ cat_note = ""
389
+
390
+ return f"Detected {len(detected_pills)} {severity} risk indicators including {', '.join(detected_pills[:3])}.{cat_note} Risk score: {risk_score}/100."
391
+
392
+ async def detect_contradictions(
393
+ self,
394
+ requirements: List[ComplianceItem]
395
+ ) -> List[Contradiction]:
396
+ """
397
+ Spec 7.2: Contradiction Detection.
398
+
399
+ Detects internal contradictions in RFP requirements using:
400
+ 1. Rule-based detection for common conflict patterns
401
+ 2. NLI-style semantic analysis for subtle contradictions
402
+ """
403
+ logger.info(f"Analyzing {len(requirements)} requirements for contradictions")
404
+
405
+ contradictions = []
406
+ contradiction_id = 0
407
+
408
+ # Group requirements by topic for pairwise comparison
409
+ topic_groups = self._group_by_topic(requirements)
410
+
411
+ # 1. Rule-based detection for page limits and format conflicts
412
+ page_limit_reqs = [r for r in requirements if self._is_page_related(r)]
413
+ format_reqs = [r for r in requirements if self._is_format_related(r)]
414
+ deadline_reqs = [r for r in requirements if self._is_deadline_related(r)]
415
+
416
+ # Check for page limit contradictions
417
+ for i, req_a in enumerate(page_limit_reqs):
418
+ for req_b in page_limit_reqs[i+1:]:
419
+ contradiction = self._check_page_limit_contradiction(req_a, req_b)
420
+ if contradiction:
421
+ contradiction_id += 1
422
+ contradiction.id = f"CONTRA-{contradiction_id:03d}"
423
+ contradictions.append(contradiction)
424
+
425
+ # Check for format contradictions
426
+ for i, req_a in enumerate(format_reqs):
427
+ for req_b in format_reqs[i+1:]:
428
+ contradiction = self._check_format_contradiction(req_a, req_b)
429
+ if contradiction:
430
+ contradiction_id += 1
431
+ contradiction.id = f"CONTRA-{contradiction_id:03d}"
432
+ contradictions.append(contradiction)
433
+
434
+ # 2. LLM-based semantic contradiction detection
435
+ if self.ai_service and len(requirements) > 1:
436
+ semantic_contradictions = await self._detect_semantic_contradictions(
437
+ requirements[:50] # Limit for API efficiency
438
+ )
439
+ for c in semantic_contradictions:
440
+ contradiction_id += 1
441
+ c.id = f"CONTRA-{contradiction_id:03d}"
442
+ contradictions.append(c)
443
+
444
+ logger.info(f"Found {len(contradictions)} potential contradictions")
445
+ return contradictions
446
+
447
+ def _is_page_related(self, req: ComplianceItem) -> bool:
448
+ """Check if requirement relates to page limits."""
449
+ keywords = ["page", "pages", "page limit", "page count", "maximum pages"]
450
+ return any(kw in req.original_text.lower() for kw in keywords)
451
+
452
+ def _is_format_related(self, req: ComplianceItem) -> bool:
453
+ """Check if requirement relates to formatting."""
454
+ keywords = ["font", "margin", "spacing", "format", "times new roman", "arial"]
455
+ return any(kw in req.original_text.lower() for kw in keywords)
456
+
457
+ def _is_deadline_related(self, req: ComplianceItem) -> bool:
458
+ """Check if requirement relates to deadlines."""
459
+ keywords = ["deadline", "due date", "submit by", "submission date", "close date"]
460
+ return any(kw in req.original_text.lower() for kw in keywords)
461
+
462
+ def _group_by_topic(self, requirements: List[ComplianceItem]) -> Dict[str, List[ComplianceItem]]:
463
+ """Group requirements by detected topic for comparison."""
464
+ topics = {
465
+ "page_limit": [],
466
+ "format": [],
467
+ "deadline": [],
468
+ "staffing": [],
469
+ "technical": [],
470
+ "pricing": [],
471
+ "other": []
472
+ }
473
+
474
+ for req in requirements:
475
+ text = req.original_text.lower()
476
+ if self._is_page_related(req):
477
+ topics["page_limit"].append(req)
478
+ elif self._is_format_related(req):
479
+ topics["format"].append(req)
480
+ elif self._is_deadline_related(req):
481
+ topics["deadline"].append(req)
482
+ elif any(kw in text for kw in ["staff", "personnel", "resume", "key person"]):
483
+ topics["staffing"].append(req)
484
+ elif any(kw in text for kw in ["price", "cost", "rate", "budget"]):
485
+ topics["pricing"].append(req)
486
+ else:
487
+ topics["other"].append(req)
488
+
489
+ return topics
490
+
491
+ def _check_page_limit_contradiction(
492
+ self,
493
+ req_a: ComplianceItem,
494
+ req_b: ComplianceItem
495
+ ) -> Optional[Contradiction]:
496
+ """Check for contradictions between two page-related requirements."""
497
+ # Extract page numbers from requirements
498
+ nums_a = re.findall(r'(\d+)\s*pages?', req_a.original_text.lower())
499
+ nums_b = re.findall(r'(\d+)\s*pages?', req_b.original_text.lower())
500
+
501
+ if nums_a and nums_b:
502
+ num_a = int(nums_a[0])
503
+ num_b = int(nums_b[0])
504
+
505
+ # Check if one is significantly different from another
506
+ if abs(num_a - num_b) > 5 and min(num_a, num_b) > 0:
507
+ # Potential contradiction - different page limits
508
+ return Contradiction(
509
+ id="",
510
+ clause_a=req_a.original_text[:200],
511
+ clause_b=req_b.original_text[:200],
512
+ clause_a_location=f"Page {req_a.page_number}",
513
+ clause_b_location=f"Page {req_b.page_number}",
514
+ contradiction_type=ContradictionType.PAGE_LIMIT,
515
+ severity="high" if abs(num_a - num_b) > 10 else "medium",
516
+ description=f"Conflicting page limits: {num_a} pages vs {num_b} pages",
517
+ recommendation="Submit RFI to clarify the correct page limit requirement"
518
+ )
519
+
520
+ return None
521
+
522
+ def _check_format_contradiction(
523
+ self,
524
+ req_a: ComplianceItem,
525
+ req_b: ComplianceItem
526
+ ) -> Optional[Contradiction]:
527
+ """Check for contradictions between two format-related requirements."""
528
+ # Check for conflicting font specifications
529
+ fonts_a = re.findall(r'(times new roman|arial|calibri|courier)', req_a.original_text.lower())
530
+ fonts_b = re.findall(r'(times new roman|arial|calibri|courier)', req_b.original_text.lower())
531
+
532
+ if fonts_a and fonts_b and fonts_a[0] != fonts_b[0]:
533
+ return Contradiction(
534
+ id="",
535
+ clause_a=req_a.original_text[:200],
536
+ clause_b=req_b.original_text[:200],
537
+ clause_a_location=f"Page {req_a.page_number}",
538
+ clause_b_location=f"Page {req_b.page_number}",
539
+ contradiction_type=ContradictionType.FORMAT,
540
+ severity="medium",
541
+ description=f"Conflicting font requirements: {fonts_a[0]} vs {fonts_b[0]}",
542
+ recommendation="Submit RFI to clarify the required font"
543
+ )
544
+
545
+ # Check for conflicting font sizes
546
+ sizes_a = re.findall(r'(\d+)\s*(?:pt|point)', req_a.original_text.lower())
547
+ sizes_b = re.findall(r'(\d+)\s*(?:pt|point)', req_b.original_text.lower())
548
+
549
+ if sizes_a and sizes_b and sizes_a[0] != sizes_b[0]:
550
+ return Contradiction(
551
+ id="",
552
+ clause_a=req_a.original_text[:200],
553
+ clause_b=req_b.original_text[:200],
554
+ clause_a_location=f"Page {req_a.page_number}",
555
+ clause_b_location=f"Page {req_b.page_number}",
556
+ contradiction_type=ContradictionType.FORMAT,
557
+ severity="low",
558
+ description=f"Conflicting font sizes: {sizes_a[0]}pt vs {sizes_b[0]}pt",
559
+ recommendation="Use the more conservative (larger) font size"
560
+ )
561
+
562
+ return None
563
+
564
+ async def _detect_semantic_contradictions(
565
+ self,
566
+ requirements: List[ComplianceItem]
567
+ ) -> List[Contradiction]:
568
+ """Use LLM for NLI-style contradiction detection."""
569
+ if not self.ai_service or len(requirements) < 2:
570
+ return []
571
+
572
+ # Prepare requirements text for analysis
573
+ req_texts = [
574
+ f"[Page {r.page_number}, {r.section_type.value if r.section_type else 'General'}]: {r.original_text[:300]}"
575
+ for r in requirements[:30] # Limit for token efficiency
576
+ ]
577
+
578
+ prompt = f"""Analyze these RFP requirements for internal contradictions. Look for:
579
+
580
+ 1. Page limits that conflict with content requirements
581
+ 2. Deadlines that conflict with each other
582
+ 3. Format requirements that contradict
583
+ 4. Scope descriptions that conflict
584
+ 5. Technical requirements that are mutually exclusive
585
+
586
+ Requirements:
587
+ {chr(10).join(req_texts)}
588
+
589
+ For each contradiction found, provide in JSON format:
590
+ {{
591
+ "contradictions": [
592
+ {{
593
+ "clause_a_idx": <index of first clause>,
594
+ "clause_b_idx": <index of second clause>,
595
+ "type": "page_limit" | "format" | "deadline" | "requirement" | "scope",
596
+ "severity": "high" | "medium" | "low",
597
+ "description": "Brief description of the conflict",
598
+ "recommendation": "How to resolve or clarify"
599
+ }}
600
+ ]
601
+ }}
602
+
603
+ Return empty array if no contradictions found. Only flag clear contradictions, not minor inconsistencies."""
604
+
605
+ try:
606
+ messages = [
607
+ {
608
+ "role": "system",
609
+ "content": "You are an RFP analyst expert at identifying conflicting requirements. Be precise and only flag actual contradictions."
610
+ },
611
+ {"role": "user", "content": prompt}
612
+ ]
613
+
614
+ chat_messages = [ChatMessage(role=m["role"], content=m["content"]) for m in messages]
615
+
616
+ response = await self.ai_service.chat(
617
+ messages=chat_messages,
618
+ feature="contradictions",
619
+ temperature=0.2
620
+ )
621
+
622
+ content = response.content
623
+ # Clean up JSON
624
+ if "```json" in content:
625
+ content = content.split("```json")[1].split("```")[0].strip()
626
+ elif "```" in content:
627
+ content = content.split("```")[1].split("```")[0].strip()
628
+
629
+ result = json.loads(content)
630
+ contradictions = []
631
+
632
+ for c in result.get("contradictions", []):
633
+ idx_a = c.get("clause_a_idx", 0)
634
+ idx_b = c.get("clause_b_idx", 1)
635
+
636
+ if idx_a < len(requirements) and idx_b < len(requirements):
637
+ req_a = requirements[idx_a]
638
+ req_b = requirements[idx_b]
639
+
640
+ contradictions.append(Contradiction(
641
+ id="",
642
+ clause_a=req_a.original_text[:200],
643
+ clause_b=req_b.original_text[:200],
644
+ clause_a_location=f"Page {req_a.page_number}",
645
+ clause_b_location=f"Page {req_b.page_number}",
646
+ contradiction_type=ContradictionType(c.get("type", "requirement")),
647
+ severity=c.get("severity", "medium"),
648
+ description=c.get("description", "Potential conflict detected"),
649
+ recommendation=c.get("recommendation", "Review and clarify with contracting officer")
650
+ ))
651
+
652
+ return contradictions
653
+
654
+ except Exception as e:
655
+ logger.error(f"Semantic contradiction detection failed: {e}")
656
+ return []
657
+
658
+ async def analyze_blind_rfp(
659
+ self,
660
+ requirements: List[ComplianceItem],
661
+ rfp_text: Optional[str] = None
662
+ ) -> Dict[str, Any]:
663
+ """
664
+ Spec 4.2: Detect 'Blind/Wired RFP' indicators.
665
+
666
+ A Blind RFP is one written with a specific competitor in mind,
667
+ making a win statistically improbable.
668
+ """
669
+ logger.info("Analyzing RFP for blind/wired indicators")
670
+
671
+ indicators: List[BlindRFPIndicator] = []
672
+ blind_score = 0
673
+
674
+ # 1. Pattern-based detection
675
+ for req in requirements:
676
+ text = req.original_text
677
+
678
+ # Check vendor-specific patterns
679
+ for pattern in self.blind_rfp_patterns["vendor_specific"]:
680
+ if re.search(pattern, text, re.IGNORECASE):
681
+ indicators.append(BlindRFPIndicator(
682
+ indicator_type="vendor_specific",
683
+ description="Potentially vendor-specific requirement detected",
684
+ evidence=text[:200],
685
+ confidence=0.7
686
+ ))
687
+ blind_score += 15
688
+
689
+ # Check narrow qualification patterns
690
+ for pattern in self.blind_rfp_patterns["narrow_qualifications"]:
691
+ if re.search(pattern, text, re.IGNORECASE):
692
+ indicators.append(BlindRFPIndicator(
693
+ indicator_type="narrow_qualifications",
694
+ description="Unusually narrow qualification requirement",
695
+ evidence=text[:200],
696
+ confidence=0.6
697
+ ))
698
+ blind_score += 10
699
+
700
+ # Check unrealistic timeline patterns
701
+ for pattern in self.blind_rfp_patterns["unrealistic_timeline"]:
702
+ if re.search(pattern, text, re.IGNORECASE):
703
+ indicators.append(BlindRFPIndicator(
704
+ indicator_type="unrealistic_timeline",
705
+ description="Unrealistic timeline suggesting incumbent advantage",
706
+ evidence=text[:200],
707
+ confidence=0.5
708
+ ))
709
+ blind_score += 10
710
+
711
+ # 2. LLM-based semantic analysis for blind RFP
712
+ if self.ai_service and rfp_text:
713
+ llm_analysis = await self._llm_blind_rfp_analysis(rfp_text[:8000])
714
+ if llm_analysis:
715
+ blind_score = int(blind_score * 0.4 + llm_analysis.get("blind_probability", 0) * 0.6)
716
+ for indicator in llm_analysis.get("indicators", []):
717
+ indicators.append(BlindRFPIndicator(
718
+ indicator_type=indicator.get("type", "other"),
719
+ description=indicator.get("description", ""),
720
+ evidence=indicator.get("evidence", ""),
721
+ confidence=indicator.get("confidence", 0.5)
722
+ ))
723
+
724
+ # Cap score at 100
725
+ blind_score = min(blind_score, 100)
726
+
727
+ return {
728
+ "is_potentially_blind": blind_score > 40,
729
+ "blind_probability": blind_score,
730
+ "indicators": [
731
+ {
732
+ "type": i.indicator_type,
733
+ "description": i.description,
734
+ "evidence": i.evidence,
735
+ "confidence": i.confidence
736
+ }
737
+ for i in indicators
738
+ ],
739
+ "recommendation": self._get_blind_rfp_recommendation(blind_score),
740
+ "analysis_method": "hybrid" if self.ai_service else "rule-based"
741
+ }
742
+
743
+ async def _llm_blind_rfp_analysis(self, rfp_text: str) -> Optional[Dict[str, Any]]:
744
+ """Use LLM for deeper blind RFP analysis."""
745
+ if not self.ai_service:
746
+ return None
747
+
748
+ prompt = f"""Analyze this RFP for signs that it may be "wired" or written to favor a specific incumbent or competitor.
749
+
750
+ Look for:
751
+ 1. Unusually specific technical requirements that only one vendor could meet
752
+ 2. References to proprietary systems without allowing equivalents
753
+ 3. Unrealistic timelines that favor incumbents
754
+ 4. Overly narrow qualification requirements
755
+ 5. Language suggesting prior relationship expectations
756
+
757
+ RFP Excerpt:
758
+ {rfp_text[:6000]}
759
+
760
+ Provide analysis in JSON:
761
+ {{
762
+ "blind_probability": <0-100>,
763
+ "indicators": [
764
+ {{
765
+ "type": "vendor_specific" | "narrow_qualifications" | "incumbent_advantage" | "proprietary_reference",
766
+ "description": "what makes this suspicious",
767
+ "evidence": "specific text or requirement",
768
+ "confidence": <0-1>
769
+ }}
770
+ ],
771
+ "overall_assessment": "brief summary"
772
+ }}"""
773
+
774
+ try:
775
+ messages = [
776
+ {"role": "system", "content": "You are an expert at analyzing government RFPs for fairness and competitiveness."},
777
+ {"role": "user", "content": prompt}
778
+ ]
779
+
780
+ chat_messages = [ChatMessage(role=m["role"], content=m["content"]) for m in messages]
781
+
782
+ response = await self.ai_service.chat(
783
+ messages=chat_messages,
784
+ feature="blind_rfp",
785
+ temperature=0.3
786
+ )
787
+
788
+ content = response.content
789
+ # Clean up JSON
790
+ if "```json" in content:
791
+ content = content.split("```json")[1].split("```")[0].strip()
792
+ elif "```" in content:
793
+ content = content.split("```")[1].split("```")[0].strip()
794
+
795
+ return json.loads(content)
796
+ except Exception as e:
797
+ logger.error(f"LLM blind RFP analysis failed: {e}")
798
+ return None
799
+
800
+ def _get_blind_rfp_recommendation(self, score: int) -> str:
801
+ """Get recommendation based on blind RFP score."""
802
+ if score > 70:
803
+ return "HIGH RISK: This RFP shows strong indicators of being wired. Consider No-Go unless there's a strategic reason to bid. If bidding, submit RFIs to broaden requirements."
804
+ elif score > 40:
805
+ return "MODERATE RISK: Some concerning indicators present. Submit RFIs to clarify narrow requirements before committing significant resources."
806
+ else:
807
+ return "LOW RISK: No significant blind RFP indicators detected. Proceed with standard bid process."
808
+
809
+ async def generate_win_themes(
810
+ self,
811
+ requirements: List[ComplianceItem],
812
+ company_strengths: Optional[List[str]] = None,
813
+ competitor_weaknesses: Optional[List[str]] = None
814
+ ) -> List[Dict[str, Any]]:
815
+ """
816
+ Generate win themes based on requirements and company positioning.
817
+ """
818
+ if not self.ai_service:
819
+ return []
820
+
821
+ # Group requirements by volume
822
+ by_volume = {}
823
+ for req in requirements:
824
+ vol = req.volume_assignment.value
825
+ if vol not in by_volume:
826
+ by_volume[vol] = []
827
+ by_volume[vol].append(req.semantic_summary or req.original_text[:100])
828
+
829
+ prompt = f"""Based on these RFP requirements grouped by volume, suggest 4-5 compelling win themes:
830
+
831
+ Requirements by Volume:
832
+ {json.dumps(by_volume, indent=2)}
833
+
834
+ Company Strengths:
835
+ {json.dumps(company_strengths or ['Not specified'])}
836
+
837
+ Competitor Weaknesses (to exploit):
838
+ {json.dumps(competitor_weaknesses or ['Not specified'])}
839
+
840
+ For each theme, provide:
841
+ 1. Theme title (catchy, memorable)
842
+ 2. Brief description (2-3 sentences)
843
+ 3. Which requirements it directly addresses
844
+ 4. Suggested proof points/evidence
845
+ 5. Discriminator level (strong/moderate/weak)
846
+
847
+ Respond in JSON format:
848
+ {{"themes": [
849
+ {{
850
+ "title": "...",
851
+ "description": "...",
852
+ "addresses": ["req1", "req2"],
853
+ "proof_points": ["evidence1", "evidence2"],
854
+ "discriminator_level": "strong|moderate|weak"
855
+ }}
856
+ ]}}"""
857
+
858
+ try:
859
+ messages = [
860
+ ChatMessage(role="system", content="You are a proposal strategist expert at developing compelling win themes that differentiate from competition."),
861
+ ChatMessage(role="user", content=prompt)
862
+ ]
863
+
864
+ response = await self.ai_service.chat(
865
+ messages=messages,
866
+ feature="win_themes",
867
+ temperature=0.5
868
+ )
869
+
870
+ content = response.content
871
+ # Clean up JSON
872
+ if "```json" in content:
873
+ content = content.split("```json")[1].split("```")[0].strip()
874
+ elif "```" in content:
875
+ content = content.split("```")[1].split("```")[0].strip()
876
+
877
+ result = json.loads(content)
878
+ return result.get("themes", [])
879
+
880
+ except Exception as e:
881
+ logger.error(f"Win theme generation failed: {e}")
882
+ return []
883
+
884
+ def calculate_compliance_health(self, requirements: List[ComplianceItem]) -> Dict[str, Any]:
885
+ """
886
+ Calculate overall compliance health metrics.
887
+ """
888
+ total = len(requirements)
889
+ if total == 0:
890
+ return {"health_score": 100, "stats": {}}
891
+
892
+ high_risk = sum(1 for r in requirements if r.risk_score >= 0.7)
893
+ medium_risk = sum(1 for r in requirements if 0.3 <= r.risk_score < 0.7)
894
+ low_risk = total - high_risk - medium_risk
895
+
896
+ # Health score: penalize high-risk items more
897
+ health_score = 100 - (high_risk * 5 + medium_risk * 2)
898
+ health_score = max(0, min(100, health_score))
899
+
900
+ # Count by type and volume
901
+ by_type = {}
902
+ by_volume = {}
903
+ for r in requirements:
904
+ t = r.compliance_type.value
905
+ v = r.volume_assignment.value
906
+ by_type[t] = by_type.get(t, 0) + 1
907
+ by_volume[v] = by_volume.get(v, 0) + 1
908
+
909
+ return {
910
+ "health_score": health_score,
911
+ "total_requirements": total,
912
+ "high_risk_count": high_risk,
913
+ "medium_risk_count": medium_risk,
914
+ "low_risk_count": low_risk,
915
+ "risk_distribution": {
916
+ "high": round(high_risk / total * 100, 1) if total else 0,
917
+ "medium": round(medium_risk / total * 100, 1) if total else 0,
918
+ "low": round(low_risk / total * 100, 1) if total else 0
919
+ },
920
+ "by_compliance_type": by_type,
921
+ "by_volume": by_volume
922
+ }
backend/app/assembly/__init__.py ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ # Assembly Engine module
2
+ from .service import AssemblyEngine
backend/app/assembly/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (207 Bytes). View file
 
backend/app/assembly/__pycache__/service.cpython-313.pyc ADDED
Binary file (43.1 kB). View file
 
backend/app/assembly/service.py ADDED
@@ -0,0 +1,939 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Assembly Engine - Comprehensive Document Generation Module
3
+
4
+ Implements:
5
+ - 4.1: Automated Excel Matrix Generation with Auto-Grouping, Data Validation, Conditional Formatting
6
+ - 6.1: Advanced Word Automation with Style Management
7
+ - 6.2: Automated PDF Form Filling with Field Mapping and Flattening
8
+ """
9
+
10
+ from docxtpl import DocxTemplate
11
+ import openpyxl
12
+ from openpyxl.styles import Font, Alignment, PatternFill, Border, Side, NamedStyle
13
+ from openpyxl.utils import get_column_letter
14
+ from openpyxl.utils.dataframe import dataframe_to_rows
15
+ from openpyxl.worksheet.datavalidation import DataValidation
16
+ from openpyxl.formatting.rule import ColorScaleRule, CellIsRule, FormulaRule
17
+ from openpyxl.chart import PieChart, Reference
18
+ import pandas as pd
19
+ import fitz # PyMuPDF
20
+ import logging
21
+ from typing import Dict, Any, List, Optional, Tuple
22
+ from pathlib import Path
23
+ from docx import Document
24
+ from docx.shared import Inches, Pt, Cm, RGBColor
25
+ from docx.enum.text import WD_ALIGN_PARAGRAPH
26
+ from docx.enum.style import WD_STYLE_TYPE
27
+ from docx.enum.table import WD_TABLE_ALIGNMENT
28
+ import os
29
+ import json
30
+ import re
31
+ from datetime import datetime
32
+
33
+ from app.core.config import settings
34
+ from app.shredding.service import ComplianceItem, ComplianceType, VolumeAssignment, SectionType
35
+
36
+ logger = logging.getLogger(__name__)
37
+
38
+
39
+ class AssemblyEngine:
40
+ """
41
+ Assembly Engine for generating proposal documents.
42
+ Supports Word, Excel, PDF form filling, and compliance matrix generation.
43
+
44
+ Implements industry-standard features:
45
+ - Auto-grouping by RFP Section
46
+ - Data Validation (dropdown menus)
47
+ - Conditional Formatting for risk visualization
48
+ - Style Management for Word documents
49
+ - PDF Form Field Mapping and Flattening
50
+ """
51
+
52
+ # Status options for data validation dropdown
53
+ STATUS_OPTIONS = ["Not Started", "In Progress", "Compliant", "Partial", "Non-Compliant", "Exception", "N/A"]
54
+
55
+ # Owner/Assignee options (can be customized per project)
56
+ DEFAULT_OWNERS = ["Technical Lead", "Subject Matter Expert", "Pricing Analyst", "Past Performance Lead", "Contracts", "Unassigned"]
57
+
58
+ # Risk keywords for conditional formatting
59
+ HIGH_RISK_KEYWORDS = [
60
+ "unlimited liability", "liquidated damages", "indemnification",
61
+ "source code escrow", "intellectual property transfer", "termination",
62
+ "performance bond", "consequential damages", "personal liability"
63
+ ]
64
+
65
+ FORMATTING_KEYWORDS = [
66
+ "font", "margin", "page limit", "format", "spacing", "header",
67
+ "footer", "table of contents", "binding", "submission"
68
+ ]
69
+
70
+ def __init__(self):
71
+ self.template_dir = Path(settings.TEMPLATE_DIR)
72
+ self.output_dir = Path(settings.OUTPUT_DIR)
73
+
74
+ # Ensure directories exist
75
+ self.template_dir.mkdir(parents=True, exist_ok=True)
76
+ self.output_dir.mkdir(parents=True, exist_ok=True)
77
+
78
+ # Define Excel styles
79
+ self._setup_excel_styles()
80
+
81
+ def _setup_excel_styles(self):
82
+ """Setup reusable Excel styles."""
83
+ self.styles = {
84
+ 'header': {
85
+ 'font': Font(bold=True, color="FFFFFF", size=11, name='Calibri'),
86
+ 'fill': PatternFill(start_color="1F4E79", end_color="1F4E79", fill_type="solid"),
87
+ 'alignment': Alignment(horizontal='center', vertical='center', wrap_text=True),
88
+ 'border': Border(
89
+ left=Side(style='thin', color='FFFFFF'),
90
+ right=Side(style='thin', color='FFFFFF'),
91
+ top=Side(style='thin', color='FFFFFF'),
92
+ bottom=Side(style='medium', color='1F4E79')
93
+ )
94
+ },
95
+ 'section_header': {
96
+ 'font': Font(bold=True, color="1F4E79", size=10, name='Calibri'),
97
+ 'fill': PatternFill(start_color="D6E3F8", end_color="D6E3F8", fill_type="solid"),
98
+ 'alignment': Alignment(horizontal='left', vertical='center'),
99
+ },
100
+ 'high_risk': {
101
+ 'fill': PatternFill(start_color="FFCDD2", end_color="FFCDD2", fill_type="solid"),
102
+ 'font': Font(color="B71C1C", name='Calibri', size=10)
103
+ },
104
+ 'medium_risk': {
105
+ 'fill': PatternFill(start_color="FFF9C4", end_color="FFF9C4", fill_type="solid"),
106
+ 'font': Font(color="F57F17", name='Calibri', size=10)
107
+ },
108
+ 'low_risk': {
109
+ 'fill': PatternFill(start_color="C8E6C9", end_color="C8E6C9", fill_type="solid"),
110
+ 'font': Font(color="1B5E20", name='Calibri', size=10)
111
+ },
112
+ 'formatting_row': {
113
+ 'fill': PatternFill(start_color="ECEFF1", end_color="ECEFF1", fill_type="solid"),
114
+ 'font': Font(color="607D8B", name='Calibri', size=10, italic=True)
115
+ }
116
+ }
117
+
118
+ def generate_compliance_matrix_excel(
119
+ self,
120
+ requirements: List[ComplianceItem],
121
+ output_path: str,
122
+ project_name: str = "RFP Response",
123
+ owners: Optional[List[str]] = None,
124
+ include_cross_references: bool = True
125
+ ) -> Dict[str, Any]:
126
+ """
127
+ Spec 4.1: Automated Excel Matrix Generation
128
+
129
+ Features:
130
+ - Auto-Grouping: Organizes requirements by RFP Section using pandas grouping
131
+ - Data Validation: Dropdown menus for Status and Owner columns
132
+ - Conditional Formatting:
133
+ - Red highlighting for high-risk keywords
134
+ - Gray highlighting for formatting requirements
135
+ - Color-coded risk scores
136
+ - Summary Statistics sheet
137
+ """
138
+ logger.info(f"Generating compliance matrix with {len(requirements)} requirements")
139
+
140
+ if not requirements:
141
+ return {"status": "error", "message": "No requirements provided"}
142
+
143
+ try:
144
+ wb = openpyxl.Workbook()
145
+ ws = wb.active
146
+ ws.title = "Compliance Matrix"
147
+
148
+ owners = owners or self.DEFAULT_OWNERS
149
+
150
+ # Define headers
151
+ headers = [
152
+ "Req ID", "RFP Section", "Section Type", "Requirement Text",
153
+ "Compliance Type", "Volume", "Risk Score", "Risk Level",
154
+ "Assigned To", "Status", "Response/Evidence", "Notes"
155
+ ]
156
+
157
+ if include_cross_references:
158
+ headers.append("Cross References")
159
+
160
+ # Apply headers with styling
161
+ for col, header in enumerate(headers, 1):
162
+ cell = ws.cell(row=1, column=col, value=header)
163
+ cell.font = self.styles['header']['font']
164
+ cell.fill = self.styles['header']['fill']
165
+ cell.alignment = self.styles['header']['alignment']
166
+ cell.border = self.styles['header']['border']
167
+
168
+ # Freeze header row
169
+ ws.freeze_panes = 'A2'
170
+
171
+ # Group requirements by section using pandas-style grouping
172
+ grouped_reqs = self._group_requirements_by_section(requirements)
173
+
174
+ current_row = 2
175
+ section_start_rows = {} # Track section starts for grouping
176
+
177
+ for section_name, section_reqs in grouped_reqs.items():
178
+ section_start_rows[section_name] = current_row
179
+
180
+ # Add section header row
181
+ section_cell = ws.cell(row=current_row, column=1, value=f"Section: {section_name}")
182
+ section_cell.font = self.styles['section_header']['font']
183
+ section_cell.fill = self.styles['section_header']['fill']
184
+ ws.merge_cells(start_row=current_row, start_column=1, end_row=current_row, end_column=len(headers))
185
+ current_row += 1
186
+
187
+ # Add requirements for this section
188
+ for req_idx, req in enumerate(section_reqs):
189
+ row_data = self._format_requirement_row(req, req_idx, section_name, include_cross_references)
190
+
191
+ for col, value in enumerate(row_data, 1):
192
+ cell = ws.cell(row=current_row, column=col, value=value)
193
+ cell.alignment = Alignment(vertical='top', wrap_text=True)
194
+
195
+ # Apply border
196
+ cell.border = Border(
197
+ left=Side(style='thin', color='E0E0E0'),
198
+ right=Side(style='thin', color='E0E0E0'),
199
+ top=Side(style='thin', color='E0E0E0'),
200
+ bottom=Side(style='thin', color='E0E0E0')
201
+ )
202
+
203
+ # Apply conditional formatting based on content
204
+ self._apply_row_formatting(ws, current_row, req, len(headers))
205
+
206
+ current_row += 1
207
+
208
+ # Add Data Validation for Status column (column 10)
209
+ status_validation = DataValidation(
210
+ type="list",
211
+ formula1=f'"{",".join(self.STATUS_OPTIONS)}"',
212
+ allow_blank=True,
213
+ showDropDown=False
214
+ )
215
+ status_validation.error = "Please select from the dropdown list"
216
+ status_validation.errorTitle = "Invalid Status"
217
+ ws.add_data_validation(status_validation)
218
+ status_validation.add(f'J2:J{current_row}')
219
+
220
+ # Add Data Validation for Owner column (column 9)
221
+ owner_validation = DataValidation(
222
+ type="list",
223
+ formula1=f'"{",".join(owners)}"',
224
+ allow_blank=True,
225
+ showDropDown=False
226
+ )
227
+ owner_validation.error = "Please select from the dropdown list"
228
+ owner_validation.errorTitle = "Invalid Owner"
229
+ ws.add_data_validation(owner_validation)
230
+ owner_validation.add(f'I2:I{current_row}')
231
+
232
+ # Add Conditional Formatting for Risk Score column (column 7)
233
+ # Red for >= 70%, Yellow for 30-69%, Green for < 30%
234
+ red_rule = CellIsRule(
235
+ operator='greaterThanOrEqual',
236
+ formula=['0.7'],
237
+ fill=PatternFill(start_color='FFCDD2', end_color='FFCDD2', fill_type='solid')
238
+ )
239
+ yellow_rule = CellIsRule(
240
+ operator='between',
241
+ formula=['0.3', '0.69'],
242
+ fill=PatternFill(start_color='FFF9C4', end_color='FFF9C4', fill_type='solid')
243
+ )
244
+ green_rule = CellIsRule(
245
+ operator='lessThan',
246
+ formula=['0.3'],
247
+ fill=PatternFill(start_color='C8E6C9', end_color='C8E6C9', fill_type='solid')
248
+ )
249
+
250
+ ws.conditional_formatting.add(f'G2:G{current_row}', red_rule)
251
+ ws.conditional_formatting.add(f'G2:G{current_row}', yellow_rule)
252
+ ws.conditional_formatting.add(f'G2:G{current_row}', green_rule)
253
+
254
+ # Set column widths
255
+ col_widths = {
256
+ 'A': 12, # Req ID
257
+ 'B': 15, # RFP Section
258
+ 'C': 12, # Section Type
259
+ 'D': 60, # Requirement Text
260
+ 'E': 14, # Compliance Type
261
+ 'F': 14, # Volume
262
+ 'G': 12, # Risk Score
263
+ 'H': 12, # Risk Level
264
+ 'I': 18, # Assigned To
265
+ 'J': 14, # Status
266
+ 'K': 40, # Response/Evidence
267
+ 'L': 25, # Notes
268
+ 'M': 20, # Cross References
269
+ }
270
+ for col_letter, width in col_widths.items():
271
+ if col_letter <= get_column_letter(len(headers)):
272
+ ws.column_dimensions[col_letter].width = width
273
+
274
+ # Add row height for better readability
275
+ for row in range(2, current_row):
276
+ ws.row_dimensions[row].height = 45
277
+
278
+ # Add Summary sheet
279
+ self._add_summary_sheet(wb, requirements, project_name)
280
+
281
+ # Add Statistics sheet with charts
282
+ self._add_statistics_sheet(wb, requirements)
283
+
284
+ # Save workbook
285
+ Path(output_path).parent.mkdir(parents=True, exist_ok=True)
286
+ wb.save(output_path)
287
+ logger.info(f"Compliance matrix saved to {output_path}")
288
+
289
+ return {
290
+ "status": "success",
291
+ "output_path": output_path,
292
+ "total_requirements": len(requirements),
293
+ "sections": len(grouped_reqs),
294
+ "high_risk_count": sum(1 for r in requirements if r.risk_score >= 0.7),
295
+ "sheets_created": ["Compliance Matrix", "Summary", "Statistics"]
296
+ }
297
+
298
+ except Exception as e:
299
+ logger.error(f"Compliance matrix generation failed: {e}")
300
+ return {"status": "error", "message": str(e)}
301
+
302
+ def _group_requirements_by_section(self, requirements: List[ComplianceItem]) -> Dict[str, List[ComplianceItem]]:
303
+ """Group requirements by RFP section using pandas-style grouping."""
304
+ grouped = {}
305
+
306
+ for req in requirements:
307
+ section = req.section_id or "General"
308
+ if section not in grouped:
309
+ grouped[section] = []
310
+ grouped[section].append(req)
311
+
312
+ # Sort sections naturally (Section L, Section M, Section C, then others)
313
+ section_order = {'L': 0, 'M': 1, 'C': 2}
314
+ sorted_sections = sorted(
315
+ grouped.keys(),
316
+ key=lambda x: (section_order.get(x[0] if x else 'Z', 99), x)
317
+ )
318
+
319
+ return {section: grouped[section] for section in sorted_sections}
320
+
321
+ def _format_requirement_row(
322
+ self,
323
+ req: ComplianceItem,
324
+ index: int,
325
+ section: str,
326
+ include_cross_refs: bool
327
+ ) -> List[Any]:
328
+ """Format a single requirement as a row of data."""
329
+ risk_level = "High" if req.risk_score >= 0.7 else "Medium" if req.risk_score >= 0.3 else "Low"
330
+
331
+ row = [
332
+ f"REQ-{section[:3].upper()}-{index+1:03d}", # Req ID
333
+ section, # RFP Section
334
+ req.section_type.value if req.section_type else "General", # Section Type
335
+ req.original_text[:1000], # Requirement Text (truncated)
336
+ req.compliance_type.value, # Compliance Type
337
+ req.volume_assignment.value, # Volume
338
+ req.risk_score, # Risk Score (as decimal for formatting)
339
+ risk_level, # Risk Level
340
+ "Unassigned", # Assigned To
341
+ "Not Started", # Status
342
+ "", # Response/Evidence
343
+ "", # Notes
344
+ ]
345
+
346
+ if include_cross_refs:
347
+ cross_refs = ", ".join(req.cross_references) if req.cross_references else ""
348
+ row.append(cross_refs)
349
+
350
+ return row
351
+
352
+ def _apply_row_formatting(self, ws, row: int, req: ComplianceItem, num_cols: int):
353
+ """Apply conditional formatting to a row based on content analysis."""
354
+ text_lower = req.original_text.lower()
355
+
356
+ # Check for high-risk keywords
357
+ is_high_risk = any(keyword in text_lower for keyword in self.HIGH_RISK_KEYWORDS)
358
+
359
+ # Check for formatting requirements
360
+ is_formatting = any(keyword in text_lower for keyword in self.FORMATTING_KEYWORDS)
361
+
362
+ if is_high_risk and req.risk_score >= 0.7:
363
+ # Apply high-risk styling
364
+ for col in range(1, num_cols + 1):
365
+ cell = ws.cell(row=row, column=col)
366
+ cell.fill = self.styles['high_risk']['fill']
367
+ elif is_formatting:
368
+ # Apply gray formatting styling
369
+ for col in range(1, num_cols + 1):
370
+ cell = ws.cell(row=row, column=col)
371
+ cell.fill = self.styles['formatting_row']['fill']
372
+ cell.font = self.styles['formatting_row']['font']
373
+
374
+ def _add_summary_sheet(self, wb, requirements: List[ComplianceItem], project_name: str):
375
+ """Add a summary sheet with project overview."""
376
+ summary_ws = wb.create_sheet("Summary")
377
+
378
+ # Project header
379
+ summary_ws['A1'] = "RFP Compliance Matrix Summary"
380
+ summary_ws['A1'].font = Font(bold=True, size=16, color="1F4E79")
381
+ summary_ws.merge_cells('A1:D1')
382
+
383
+ summary_data = [
384
+ ["", ""],
385
+ ["Project Name:", project_name],
386
+ ["Generated Date:", datetime.now().strftime("%Y-%m-%d %H:%M:%S")],
387
+ ["Generated By:", "RFP Automation System"],
388
+ ["", ""],
389
+ ["REQUIREMENT STATISTICS", ""],
390
+ ["Total Requirements:", len(requirements)],
391
+ ["", ""],
392
+ ["By Compliance Type:", ""],
393
+ [" Mandatory:", sum(1 for r in requirements if r.compliance_type == ComplianceType.MANDATORY)],
394
+ [" Desirable:", sum(1 for r in requirements if r.compliance_type == ComplianceType.DESIRABLE)],
395
+ [" Informational:", sum(1 for r in requirements if r.compliance_type == ComplianceType.INFORMATIONAL)],
396
+ ["", ""],
397
+ ["By Risk Level:", ""],
398
+ [" High Risk (≥70%):", sum(1 for r in requirements if r.risk_score >= 0.7)],
399
+ [" Medium Risk (30-69%):", sum(1 for r in requirements if 0.3 <= r.risk_score < 0.7)],
400
+ [" Low Risk (<30%):", sum(1 for r in requirements if r.risk_score < 0.3)],
401
+ ["", ""],
402
+ ["By Volume Assignment:", ""],
403
+ [" Technical:", sum(1 for r in requirements if r.volume_assignment == VolumeAssignment.TECHNICAL)],
404
+ [" Management:", sum(1 for r in requirements if r.volume_assignment == VolumeAssignment.MANAGEMENT)],
405
+ [" Cost/Pricing:", sum(1 for r in requirements if r.volume_assignment == VolumeAssignment.COST)],
406
+ [" Past Performance:", sum(1 for r in requirements if r.volume_assignment == VolumeAssignment.PAST_PERFORMANCE)],
407
+ ]
408
+
409
+ for row_idx, (label, value) in enumerate(summary_data, 2):
410
+ cell_a = summary_ws.cell(row=row_idx, column=1, value=label)
411
+ cell_b = summary_ws.cell(row=row_idx, column=2, value=value)
412
+
413
+ if label and not label.startswith(" ") and ":" in label:
414
+ cell_a.font = Font(bold=True)
415
+ if label.isupper():
416
+ cell_a.font = Font(bold=True, size=12, color="1F4E79")
417
+
418
+ summary_ws.column_dimensions['A'].width = 30
419
+ summary_ws.column_dimensions['B'].width = 20
420
+
421
+ def _add_statistics_sheet(self, wb, requirements: List[ComplianceItem]):
422
+ """Add a statistics sheet with visual charts."""
423
+ stats_ws = wb.create_sheet("Statistics")
424
+
425
+ # Risk distribution data
426
+ stats_ws['A1'] = "Risk Distribution"
427
+ stats_ws['A1'].font = Font(bold=True, size=14)
428
+
429
+ risk_data = [
430
+ ["Risk Level", "Count"],
431
+ ["High Risk", sum(1 for r in requirements if r.risk_score >= 0.7)],
432
+ ["Medium Risk", sum(1 for r in requirements if 0.3 <= r.risk_score < 0.7)],
433
+ ["Low Risk", sum(1 for r in requirements if r.risk_score < 0.3)],
434
+ ]
435
+
436
+ for row_idx, row_data in enumerate(risk_data, 2):
437
+ for col_idx, value in enumerate(row_data, 1):
438
+ stats_ws.cell(row=row_idx, column=col_idx, value=value)
439
+
440
+ # Add pie chart for risk distribution
441
+ try:
442
+ chart = PieChart()
443
+ chart.title = "Risk Distribution"
444
+ labels = Reference(stats_ws, min_col=1, min_row=3, max_row=5)
445
+ data = Reference(stats_ws, min_col=2, min_row=2, max_row=5)
446
+ chart.add_data(data, titles_from_data=True)
447
+ chart.set_categories(labels)
448
+ chart.width = 12
449
+ chart.height = 8
450
+ stats_ws.add_chart(chart, "D2")
451
+ except Exception as e:
452
+ logger.warning(f"Could not add chart: {e}")
453
+
454
+ def generate_proposal_doc(
455
+ self,
456
+ context: Dict[str, Any],
457
+ template_path: str,
458
+ output_path: str,
459
+ formatting_rules: Optional[Dict[str, Any]] = None
460
+ ) -> Dict[str, Any]:
461
+ """
462
+ Spec 6.1: Advanced Word Document Generation using python-docx and docxtpl.
463
+
464
+ Features:
465
+ - Style Management from template
466
+ - Formatting constraint application (from Section L)
467
+ - Dynamic content injection (tables, images)
468
+ - Table of Contents support
469
+ """
470
+ logger.info(f"Rendering proposal from template {template_path}")
471
+
472
+ try:
473
+ # Check if template exists
474
+ if not Path(template_path).exists():
475
+ # Generate from scratch if no template
476
+ return self._generate_proposal_from_scratch(context, output_path, formatting_rules)
477
+
478
+ # Use DocxTemplate for Jinja2 rendering
479
+ doc = DocxTemplate(template_path)
480
+ doc.render(context)
481
+
482
+ # Apply formatting rules from Section L if provided
483
+ if formatting_rules:
484
+ self._apply_formatting_rules(doc, formatting_rules)
485
+
486
+ # Ensure output directory exists
487
+ Path(output_path).parent.mkdir(parents=True, exist_ok=True)
488
+ doc.save(output_path)
489
+
490
+ logger.info(f"Proposal saved to {output_path}")
491
+
492
+ return {
493
+ "status": "success",
494
+ "output_path": output_path,
495
+ "template_used": template_path,
496
+ "formatting_applied": bool(formatting_rules)
497
+ }
498
+
499
+ except Exception as e:
500
+ logger.error(f"Document generation failed: {e}")
501
+ return {"status": "error", "message": str(e)}
502
+
503
+ def _generate_proposal_from_scratch(
504
+ self,
505
+ context: Dict[str, Any],
506
+ output_path: str,
507
+ formatting_rules: Optional[Dict[str, Any]] = None
508
+ ) -> Dict[str, Any]:
509
+ """Generate a proposal document from scratch without a template."""
510
+ doc = Document()
511
+
512
+ # Apply formatting rules
513
+ rules = formatting_rules or {}
514
+ font_name = rules.get('font_name', 'Times New Roman')
515
+ font_size = Pt(rules.get('font_size', 12))
516
+
517
+ # Set default paragraph style
518
+ style = doc.styles['Normal']
519
+ style.font.name = font_name
520
+ style.font.size = font_size
521
+
522
+ # Add title
523
+ title = doc.add_heading(context.get('title', 'Proposal Response'), level=0)
524
+ title.alignment = WD_ALIGN_PARAGRAPH.CENTER
525
+
526
+ # Add executive summary if provided
527
+ if 'executive_summary' in context:
528
+ doc.add_heading('Executive Summary', level=1)
529
+ doc.add_paragraph(context['executive_summary'])
530
+
531
+ # Add sections
532
+ for section in context.get('sections', []):
533
+ doc.add_heading(section.get('title', 'Section'), level=1)
534
+ doc.add_paragraph(section.get('content', ''))
535
+
536
+ # Add tables if present
537
+ if 'table' in section:
538
+ self._add_table_to_doc(doc, section['table'])
539
+
540
+ # Add compliance matrix if provided
541
+ if 'compliance_matrix' in context:
542
+ doc.add_heading('Compliance Matrix', level=1)
543
+ self._add_compliance_table_to_doc(doc, context['compliance_matrix'])
544
+
545
+ doc.save(output_path)
546
+
547
+ return {
548
+ "status": "success",
549
+ "output_path": output_path,
550
+ "template_used": None,
551
+ "generated_from_scratch": True
552
+ }
553
+
554
+ def _apply_formatting_rules(self, doc, rules: Dict[str, Any]):
555
+ """Apply Section L formatting rules to document."""
556
+ # This would modify the underlying Document object
557
+ # For DocxTemplate, we need to access the docx property
558
+ try:
559
+ underlying_doc = doc.docx
560
+
561
+ # Apply font rules
562
+ if 'font_name' in rules:
563
+ for paragraph in underlying_doc.paragraphs:
564
+ for run in paragraph.runs:
565
+ run.font.name = rules['font_name']
566
+
567
+ if 'font_size' in rules:
568
+ for paragraph in underlying_doc.paragraphs:
569
+ for run in paragraph.runs:
570
+ run.font.size = Pt(rules['font_size'])
571
+
572
+ except Exception as e:
573
+ logger.warning(f"Could not apply formatting rules: {e}")
574
+
575
+ def _add_table_to_doc(self, doc: Document, table_data: Dict[str, Any]):
576
+ """Add a table to the document."""
577
+ headers = table_data.get('headers', [])
578
+ rows = table_data.get('rows', [])
579
+
580
+ if not headers and not rows:
581
+ return
582
+
583
+ num_cols = len(headers) if headers else len(rows[0]) if rows else 0
584
+ num_rows = (1 if headers else 0) + len(rows)
585
+
586
+ table = doc.add_table(rows=num_rows, cols=num_cols)
587
+ table.style = 'Table Grid'
588
+
589
+ # Add headers
590
+ if headers:
591
+ for i, header in enumerate(headers):
592
+ cell = table.rows[0].cells[i]
593
+ cell.text = str(header)
594
+ cell.paragraphs[0].runs[0].bold = True
595
+
596
+ # Add data rows
597
+ start_row = 1 if headers else 0
598
+ for row_idx, row_data in enumerate(rows):
599
+ for col_idx, value in enumerate(row_data):
600
+ table.rows[start_row + row_idx].cells[col_idx].text = str(value)
601
+
602
+ def _add_compliance_table_to_doc(self, doc: Document, requirements: List[Dict[str, Any]]):
603
+ """Add a compliance matrix table to the document."""
604
+ headers = ["Req ID", "Requirement", "Status", "Response"]
605
+
606
+ table = doc.add_table(rows=1 + len(requirements), cols=len(headers))
607
+ table.style = 'Table Grid'
608
+
609
+ # Header row
610
+ for i, header in enumerate(headers):
611
+ cell = table.rows[0].cells[i]
612
+ cell.text = header
613
+ cell.paragraphs[0].runs[0].bold = True
614
+
615
+ # Data rows
616
+ for row_idx, req in enumerate(requirements, 1):
617
+ table.rows[row_idx].cells[0].text = req.get('id', '')
618
+ table.rows[row_idx].cells[1].text = req.get('text', '')[:200]
619
+ table.rows[row_idx].cells[2].text = req.get('status', 'Pending')
620
+ table.rows[row_idx].cells[3].text = req.get('response', '')
621
+
622
+ def fill_pdf_form(
623
+ self,
624
+ form_path: str,
625
+ data: Dict[str, Any],
626
+ output_path: str,
627
+ flatten: bool = True
628
+ ) -> Dict[str, Any]:
629
+ """
630
+ Spec 6.2: Automated PDF Form Filling with Field Mapping and Flattening.
631
+
632
+ Implementation Workflow:
633
+ 1. Field Mapping: Inspect PDF to extract field names
634
+ 2. Data Injection: Map fields to bid/RFP metadata
635
+ 3. Flattening: Make fields uneditable after filling
636
+ """
637
+ logger.info(f"Filling PDF form {form_path}")
638
+
639
+ try:
640
+ doc = fitz.open(form_path)
641
+ fields_filled = 0
642
+ field_mapping = []
643
+
644
+ for page_num, page in enumerate(doc):
645
+ for widget in page.widgets():
646
+ if widget.field_name:
647
+ field_info = {
648
+ "name": widget.field_name,
649
+ "page": page_num + 1,
650
+ "type": widget.field_type_string,
651
+ "filled": False
652
+ }
653
+
654
+ if widget.field_name in data:
655
+ fill_value = data[widget.field_name]
656
+
657
+ # Handle different field types
658
+ if widget.field_type_string == 'Btn':
659
+ # Checkbox or radio button
660
+ widget.field_value = bool(fill_value)
661
+ elif widget.field_type_string == 'Ch':
662
+ # Choice field (dropdown/listbox)
663
+ widget.field_value = str(fill_value)
664
+ else:
665
+ # Text field
666
+ widget.field_value = str(fill_value)
667
+
668
+ widget.update()
669
+ fields_filled += 1
670
+ field_info["filled"] = True
671
+ field_info["value"] = str(fill_value)[:50]
672
+
673
+ field_mapping.append(field_info)
674
+
675
+ # Save the filled form
676
+ Path(output_path).parent.mkdir(parents=True, exist_ok=True)
677
+
678
+ if flatten:
679
+ # Flatten by re-rendering (makes fields non-editable)
680
+ doc.save(output_path, deflate=True, garbage=4)
681
+ else:
682
+ doc.save(output_path)
683
+
684
+ doc.close()
685
+ logger.info(f"Filled PDF saved to {output_path}, {fields_filled} fields updated")
686
+
687
+ return {
688
+ "status": "success",
689
+ "output_path": output_path,
690
+ "fields_filled": fields_filled,
691
+ "total_fields": len(field_mapping),
692
+ "flattened": flatten,
693
+ "field_mapping": field_mapping
694
+ }
695
+
696
+ except Exception as e:
697
+ logger.error(f"PDF form filling failed: {e}")
698
+ return {"status": "error", "message": str(e)}
699
+
700
+ def extract_pdf_form_schema(self, form_path: str) -> Dict[str, Any]:
701
+ """
702
+ Extract the schema of form fields from a PDF.
703
+ Useful for understanding what data needs to be provided.
704
+ """
705
+ logger.info(f"Extracting form schema from {form_path}")
706
+
707
+ try:
708
+ doc = fitz.open(form_path)
709
+ schema = {
710
+ "total_pages": len(doc),
711
+ "fields": []
712
+ }
713
+
714
+ for page_num, page in enumerate(doc):
715
+ for widget in page.widgets():
716
+ if widget.field_name:
717
+ field = {
718
+ "name": widget.field_name,
719
+ "page": page_num + 1,
720
+ "type": widget.field_type_string,
721
+ "rect": list(widget.rect),
722
+ "required": False, # Would need additional logic to detect
723
+ "options": []
724
+ }
725
+
726
+ # Get options for choice fields
727
+ if widget.field_type_string == 'Ch' and widget.choice_values:
728
+ field["options"] = list(widget.choice_values)
729
+
730
+ schema["fields"].append(field)
731
+
732
+ doc.close()
733
+
734
+ return {
735
+ "status": "success",
736
+ "schema": schema
737
+ }
738
+
739
+ except Exception as e:
740
+ logger.error(f"Schema extraction failed: {e}")
741
+ return {"status": "error", "message": str(e)}
742
+
743
+ def generate_excel_pricing(
744
+ self,
745
+ pricing_data: Dict[str, Any],
746
+ template_path: str,
747
+ output_path: str
748
+ ) -> Dict[str, Any]:
749
+ """
750
+ Automated Excel Spreadsheet Generation using openpyxl.
751
+ Maps internal cost data to specific cells in the spreadsheet.
752
+ """
753
+ logger.info(f"Generating pricing sheet from {template_path}")
754
+
755
+ try:
756
+ # Load the client's provided pricing template
757
+ wb = openpyxl.load_workbook(template_path)
758
+
759
+ # pricing_data contains sheet names and cell mappings
760
+ for sheet_name, cells in pricing_data.items():
761
+ if sheet_name in wb.sheetnames:
762
+ ws = wb[sheet_name]
763
+ for cell_coord, value in cells.items():
764
+ ws[cell_coord] = value
765
+ else:
766
+ logger.warning(f"Sheet {sheet_name} not found in template")
767
+
768
+ wb.save(output_path)
769
+ logger.info(f"Pricing sheet saved to {output_path}")
770
+
771
+ return {
772
+ "status": "success",
773
+ "output_path": output_path,
774
+ "sheets_modified": list(pricing_data.keys())
775
+ }
776
+
777
+ except Exception as e:
778
+ logger.error(f"Excel generation failed: {e}")
779
+ return {"status": "error", "message": str(e)}
780
+
781
+ def generate_generic_pricing_sheet(
782
+ self,
783
+ data: List[Dict[str, Any]],
784
+ output_path: str,
785
+ sheet_name: str = "Pricing"
786
+ ) -> Dict[str, Any]:
787
+ """Generates a new Excel sheet from scratch using pandas."""
788
+ logger.info(f"Generating generic pricing sheet")
789
+
790
+ try:
791
+ df = pd.DataFrame(data)
792
+
793
+ with pd.ExcelWriter(output_path, engine='openpyxl') as writer:
794
+ df.to_excel(writer, sheet_name=sheet_name, index=False)
795
+
796
+ workbook = writer.book
797
+ worksheet = writer.sheets[sheet_name]
798
+
799
+ # Format headers
800
+ for cell in worksheet[1]:
801
+ cell.font = Font(bold=True)
802
+ cell.fill = PatternFill(start_color="1F4E79", end_color="1F4E79", fill_type="solid")
803
+ cell.font = Font(bold=True, color="FFFFFF")
804
+
805
+ # Auto-fit columns
806
+ for column in worksheet.columns:
807
+ max_length = 0
808
+ column_letter = column[0].column_letter
809
+ for cell in column:
810
+ try:
811
+ if len(str(cell.value)) > max_length:
812
+ max_length = len(str(cell.value))
813
+ except:
814
+ pass
815
+ adjusted_width = min(max_length + 2, 50)
816
+ worksheet.column_dimensions[column_letter].width = adjusted_width
817
+
818
+ return {
819
+ "status": "success",
820
+ "output_path": output_path,
821
+ "rows": len(data)
822
+ }
823
+
824
+ except Exception as e:
825
+ logger.error(f"Pricing sheet generation failed: {e}")
826
+ return {"status": "error", "message": str(e)}
827
+
828
+ def sanitize_document_metadata(
829
+ self,
830
+ file_path: str,
831
+ output_path: Optional[str] = None
832
+ ) -> Dict[str, Any]:
833
+ """Remove sensitive metadata from documents before submission."""
834
+ output_path = output_path or file_path
835
+ path = Path(file_path)
836
+
837
+ try:
838
+ if path.suffix.lower() == '.pdf':
839
+ return self._sanitize_pdf(file_path, output_path)
840
+ elif path.suffix.lower() == '.docx':
841
+ return self._sanitize_docx(file_path, output_path)
842
+ else:
843
+ return {"status": "error", "message": f"Unsupported file type: {path.suffix}"}
844
+
845
+ except Exception as e:
846
+ logger.error(f"Document sanitization failed: {e}")
847
+ return {"status": "error", "message": str(e)}
848
+
849
+ def _sanitize_pdf(self, input_path: str, output_path: str) -> Dict[str, Any]:
850
+ """Remove metadata from PDF."""
851
+ doc = fitz.open(input_path)
852
+ doc.set_metadata({})
853
+ doc.save(output_path, clean=True, deflate=True)
854
+ doc.close()
855
+
856
+ return {
857
+ "status": "success",
858
+ "output_path": output_path,
859
+ "metadata_cleared": True
860
+ }
861
+
862
+ def _sanitize_docx(self, input_path: str, output_path: str) -> Dict[str, Any]:
863
+ """Remove metadata from DOCX."""
864
+ doc = Document(input_path)
865
+ core_props = doc.core_properties
866
+ core_props.author = ""
867
+ core_props.comments = ""
868
+ core_props.last_modified_by = ""
869
+ doc.save(output_path)
870
+
871
+ return {
872
+ "status": "success",
873
+ "output_path": output_path,
874
+ "metadata_cleared": True
875
+ }
876
+
877
+ def create_submission_package(
878
+ self,
879
+ artifacts: List[Dict[str, Any]],
880
+ output_dir: str,
881
+ project_name: str
882
+ ) -> Dict[str, Any]:
883
+ """Create a submission-ready package with all artifacts."""
884
+ import zipfile
885
+
886
+ try:
887
+ package_dir = Path(output_dir) / f"{project_name}_submission"
888
+ package_dir.mkdir(parents=True, exist_ok=True)
889
+
890
+ processed_files = []
891
+ for artifact in artifacts:
892
+ src_path = artifact.get("path")
893
+ if src_path and Path(src_path).exists():
894
+ filename = Path(src_path).name
895
+ dest_path = package_dir / filename
896
+
897
+ if artifact.get("sanitize", True):
898
+ self.sanitize_document_metadata(src_path, str(dest_path))
899
+ else:
900
+ import shutil
901
+ shutil.copy(src_path, dest_path)
902
+
903
+ processed_files.append({
904
+ "original": src_path,
905
+ "packaged": str(dest_path),
906
+ "sanitized": artifact.get("sanitize", True)
907
+ })
908
+
909
+ # Create manifest
910
+ manifest = {
911
+ "project_name": project_name,
912
+ "created_date": datetime.now().isoformat(),
913
+ "files": processed_files,
914
+ "total_files": len(processed_files)
915
+ }
916
+
917
+ manifest_path = package_dir / "manifest.json"
918
+ with open(manifest_path, 'w') as f:
919
+ json.dump(manifest, f, indent=2)
920
+
921
+ # Create ZIP archive
922
+ zip_path = Path(output_dir) / f"{project_name}_submission.zip"
923
+ with zipfile.ZipFile(zip_path, 'w', zipfile.ZIP_DEFLATED) as zipf:
924
+ for file_path in package_dir.rglob('*'):
925
+ if file_path.is_file():
926
+ arcname = file_path.relative_to(package_dir)
927
+ zipf.write(file_path, arcname)
928
+
929
+ return {
930
+ "status": "success",
931
+ "package_dir": str(package_dir),
932
+ "zip_path": str(zip_path),
933
+ "files_processed": len(processed_files),
934
+ "manifest": manifest
935
+ }
936
+
937
+ except Exception as e:
938
+ logger.error(f"Submission package creation failed: {e}")
939
+ return {"status": "error", "message": str(e)}
backend/app/core/__init__.py ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ # Core configuration module
2
+ from .config import settings
backend/app/core/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (196 Bytes). View file
 
backend/app/core/__pycache__/config.cpython-313.pyc ADDED
Binary file (2.57 kB). View file
 
backend/app/core/config.py ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from pydantic_settings import BaseSettings
2
+ from typing import Optional
3
+ import os
4
+
5
+ class Settings(BaseSettings):
6
+ PROJECT_NAME: str = "Next-Gen RFP Automation"
7
+ API_V1_STR: str = "/api/v1"
8
+
9
+ # AI Providers
10
+ OPENAI_API_KEY: str = ""
11
+ OPENAI_BASE_URL: str = "https://api.openai.com/v1"
12
+ REAL_OPENAI_API_KEY: Optional[str] = None
13
+ OPENROUTER_API_KEY: Optional[str] = None
14
+
15
+
16
+ FIRECRAWL_API_KEY: Optional[str] = None
17
+ MISTRAL_API_KEY: Optional[str] = None
18
+ GROQ_API_KEY: Optional[str] = None
19
+ CEREBRAS_API_KEY: Optional[str] = None # Free tier for fast inference
20
+ INNGEST_EVENT_KEY: Optional[str] = None
21
+ INNGEST_SIGNING_KEY: Optional[str] = None
22
+
23
+ # Database Settings
24
+ DATABASE_URL: Optional[str] = None
25
+ ASYNC_DATABASE_URL: Optional[str] = None
26
+
27
+ # Competitor Database
28
+ COMPETITOR_DB_PATH: str = "./data/competitors.json"
29
+
30
+ # NLI Model Settings
31
+ USE_TRANSFORMER_NLI: bool = False # Set to True if transformers installed
32
+ CROSS_ENCODER_MODEL: str = "balanced" # "high_accuracy", "balanced", or "fast"
33
+
34
+ # ChromaDB settings
35
+ CHROMA_PERSIST_DIR: str = "./chroma_db"
36
+
37
+ # Embedding settings
38
+ EMBEDDING_MODEL: str = "text-embedding-3-small"
39
+ EMBEDDING_DIMENSIONS: int = 1536
40
+
41
+ # Chunking settings
42
+ CHUNK_SIZE: int = 512
43
+ CHUNK_OVERLAP: int = 50
44
+
45
+ # Storage paths
46
+ UPLOAD_DIR: str = "./uploads"
47
+ OUTPUT_DIR: str = "./outputs"
48
+ TEMPLATE_DIR: str = "./templates"
49
+
50
+ class Config:
51
+ env_file = ".env"
52
+ case_sensitive = True
53
+ extra = "ignore" # Ignore extra env vars not defined in the model
54
+
55
+ settings = Settings()
56
+
57
+ # Ensure directories exist
58
+ for dir_path in [settings.UPLOAD_DIR, settings.OUTPUT_DIR, settings.TEMPLATE_DIR, settings.CHROMA_PERSIST_DIR, "./data"]:
59
+ os.makedirs(dir_path, exist_ok=True)
backend/app/database.py ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from sqlalchemy import create_engine
2
+ from sqlalchemy.ext.asyncio import create_async_engine, AsyncSession
3
+ from sqlalchemy.orm import sessionmaker, Session
4
+ from sqlalchemy.ext.declarative import declarative_base
5
+ import os
6
+ from dotenv import load_dotenv
7
+
8
+ load_dotenv()
9
+
10
+ DATABASE_URL = os.getenv("DATABASE_URL")
11
+ ASYNC_DATABASE_URL = os.getenv("ASYNC_DATABASE_URL")
12
+
13
+ # Synchronous Engine
14
+ engine = create_engine(DATABASE_URL)
15
+ SessionLocal = sessionmaker(autocommit=False, autoflush=False, bind=engine)
16
+
17
+ # Asynchronous Engine (for FastAPI async ops)
18
+ async_engine = create_async_engine(ASYNC_DATABASE_URL, echo=False)
19
+ AsyncSessionLocal = sessionmaker(
20
+ async_engine, class_=AsyncSession, expire_on_commit=False
21
+ )
22
+
23
+ def get_db():
24
+ db = SessionLocal()
25
+ try:
26
+ yield db
27
+ finally:
28
+ db.close()
29
+
30
+ async def get_async_db():
31
+ async with AsyncSessionLocal() as session:
32
+ yield session
backend/app/ingestion/__init__.py ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ # Ingestion Engine module
2
+ from .service import IngestionService, IngestedPage, DocumentStructure