-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathknowledge-base.js
More file actions
341 lines (335 loc) · 21.2 KB
/
Copy pathknowledge-base.js
File metadata and controls
341 lines (335 loc) · 21.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
// ─── Knowledge Base ───
// Sourced from career-garage Resume-Factory (Sep 2026)
const knowledgeBase = {
about: {
name: 'Arijit Kumar Roy',
role: 'Senior Software Engineer, Technical Lead — Data & AI Platform Engineering',
headline: 'Data & Strategy Lead for Voice AI / Voice Agents',
experience: '8 years',
location: 'Bangalore, India',
email: 'arijitroy003@gmail.com',
github: 'arijitroy003',
linkedin: 'sudo-kill',
website: 'https://arijitroy003.github.io'
},
experience: [
{
company: 'Red Hat',
role: 'Senior Software Engineer, Technical Lead — Data & AI Platform Engineering',
period: 'Apr 2024 — Present',
highlights: [
'First hire of the Data & AI Platform team; leading 5 engineers on production agents',
'On-Call / Data Reliability Agent (MCP, LangChain, Langfuse) cut investigation MTTR from ~38 minutes to ~4 minutes',
'Data contracts and catalog completeness for 150+ data products; adoption grew from ~30 to 730 users',
'AI-assisted GitLab merge-request reviewer on Vertex AI with LLM evaluation',
'Release Assistant saved 1000+ lead engineer hours annually',
'GitOps data plane on OpenShift (Python, Golang, Kubernetes); $200k+ platform cost reduction from legacy Redshift/Starburst'
]
},
{
company: 'Beem',
role: 'Senior Data Engineer — Financial Services',
period: 'Nov 2023 — Mar 2024',
highlights: [
'LLM-powered data and AI platform serving 50M+ users for personal finance',
'Data insights for investor pitches that secured $16k Databricks funding with $24k in future commitments',
'500 GB/day clickstream pipelines on AWS S3, Python, Databricks, and Mixpanel'
]
},
{
company: 'Tata Digital (Tata Neu)',
role: 'Senior Software Engineer — E-Commerce, Retail',
period: 'Aug 2021 — Oct 2023',
highlights: [
'Conversational AI (voice and text) platform serving 120M+ users and 500M events/day',
'Voice-call analysis with speech-to-text, audio ML, and custom voice models',
'Monitored an AI chatbot across 12 Indic languages',
'Quality metrics: unique/repeat users, response time, intent and escalation funnels; 24x7 Kusto dashboards',
'Led 6 engineers on 15+ customer-facing pipelines; lakehouse migration cut latency 75% and cost 80%',
'GenAI product search with Azure OpenAI, LangChain, embeddings, and vector databases'
]
},
{
company: 'Gnosis Lab',
role: 'Founding Engineer — NASSCOM 10K Startups',
period: 'Jun 2019 — May 2021',
highlights: [
'Founding engineer for 0-to-1 SaaS backends on AWS',
'Social-marketing automation, an AI marketing bot, and an LMS delivering 50+ APIs'
]
}
],
skills: {
voice: ['Conversational AI', 'STT', 'Audio ML', 'Multilingual (12 Indic)', 'Call analytics', 'Intent / escalation funnels'],
agents: ['LangChain', 'MCP', 'Langfuse', 'LLM-as-judge', 'Tool / function calling', 'Cursor', 'Claude Code'],
data: ['SQL', 'Data quality', 'Data contracts', 'Metric definitions', 'Collection pipelines', 'Catalog / semantic layer', 'Atlan', 'Snowflake', 'Databricks', 'PySpark', 'Delta Lake', 'dbt', 'Airflow', 'Kafka', 'Fivetran'],
platform: ['Python', 'Golang', 'Kubernetes', 'OpenShift', 'GitOps', 'AWS', 'Azure', 'GCP', 'Observability'],
languages: ['Python', 'Golang', 'SQL', 'JavaScript', 'TypeScript', 'Scala', 'Shell'],
spoken: ['English', 'Bengali', 'Hindi']
},
education: [
{ school: 'Jadavpur University', degree: 'Masters in Computer Applications', focus: 'Distributed Systems / Cloud Computing', gpa: '8.81/10', period: '2018-2021' },
{ school: 'Ramakrishna Mission Vidyamandira', degree: 'B.Sc. Computer Science', gpa: '8.49/10', period: '2015-2018' }
],
projects: [
'On-Call / Data Reliability Agent at Red Hat (MCP, LangChain, Langfuse)',
'GitOps Data Mesh and catalog plane at Red Hat (150+ data products)',
'Conversational AI voice and text platform at Tata Neu (120M+ users)',
'Voice-call analysis and 12-language chatbot monitoring at Tata Neu',
'GenAI product search at Tata Neu',
'LLM-powered personal finance platform at Beem',
'Doctor — clinical management app (React Native, GCP, Firebase)'
]
};
const responses = {
greeting: [
'Hello. I am Arijit Kumar Roy’s portfolio assistant. I can walk you through his experience, projects, and how to get in touch.',
'Welcome. I can help you learn about Arijit’s work in voice AI, data platforms, and agentic systems. What would you like to start with?'
],
experience: () => {
const exp = knowledgeBase.experience;
return (
`Arijit has **${knowledgeBase.about.experience}** of experience across voice AI, data platforms, and production agents.\n\n` +
`He is currently **${exp[0].role}** at **${exp[0].company}** — first hire of the Data & AI Platform team, leading five engineers.\n\n` +
`Before that: **${exp[1].company}** (LLM finance platform, 50M+ users), **${exp[2].company}** (conversational AI for 120M+ users), and founding engineer at **${exp[3].company}** (NASSCOM 10K).`
);
},
redhat: () => (
`**Red Hat** (Apr 2024 — Present) — Senior Software Engineer, Technical Lead, Data & AI Platform Engineering.\n\n` +
`• First hire of the team; leads 5 engineers on production agents\n` +
`• On-Call / Data Reliability Agent (MCP, LangChain, Langfuse) reduced investigation MTTR from ~38 minutes to ~4 minutes\n` +
`• Data contracts and catalog completeness for 150+ data products; adoption grew from ~30 to 730 users\n` +
`• GitLab merge-request reviewer on Vertex AI with LLM evaluation\n` +
`• Release Assistant saved 1000+ lead engineer hours a year\n` +
`• GitOps data plane on OpenShift; $200k+ platform cost reduction from legacy Redshift/Starburst`
),
tata: () => (
`**Tata Digital / Tata Neu** (Aug 2021 — Oct 2023) — Senior Software Engineer, Strategic Initiatives.\n\n` +
`• Conversational AI (voice and text) for **120M+ users**, **500M events/day**\n` +
`• Voice-call analysis with speech-to-text, audio ML, and custom voice models\n` +
`• Chatbot monitoring across **12 Indic languages**\n` +
`• Quality metrics: unique/repeat users, response time, intent and escalation funnels; 24x7 Kusto dashboards\n` +
`• Led 6 engineers; lakehouse migration cut latency **75%** and cost **80%**\n` +
`• GenAI product search with Azure OpenAI, LangChain, embeddings, and vector databases`
),
beem: () => (
`**Beem** (Nov 2023 — Mar 2024) — Senior Data Engineer, financial services.\n\n` +
`• LLM-powered data and AI platform serving **50M+ users** for personal finance\n` +
`• Contributed data insights to investor pitches that secured **$16k** Databricks funding with **$24k** in future commitments\n` +
`• Built **500 GB/day** clickstream pipelines on AWS S3, Python, Databricks, and Mixpanel`
),
gnosis: () => (
`**Gnosis Lab** (Jun 2019 — May 2021) — Founding Engineer, NASSCOM 10K Startups.\n\n` +
`• 0-to-1 SaaS backends on AWS (Python, Lambda, DynamoDB, MongoDB)\n` +
`• Social-marketing automation, an AI marketing bot, and an LMS with 50+ APIs\n` +
`This is where he learned end-to-end ownership as a first technical hire.`
),
skills: () => {
const s = knowledgeBase.skills;
return (
`**Voice / speech:** ${s.voice.join(', ')}\n\n` +
`**Agents / evals:** ${s.agents.join(', ')}\n\n` +
`**Data & strategy:** ${s.data.slice(0, 8).join(', ')}\n\n` +
`**Platform:** ${s.platform.join(', ')}\n\n` +
`**Languages:** ${s.languages.join(', ')}`
);
},
ai: () => (
`Arijit’s AI work is production-facing, not demo-only.\n\n` +
`• Conversational AI at Tata Neu: voice + text, 120M+ users, speech/STT/audio ML across 12 Indic languages\n` +
`• Agents at Red Hat: MCP, LangChain, Langfuse, and LLM-as-judge evaluation\n` +
`• On-Call / Data Reliability Agent cut MTTR from ~38 minutes to ~4 minutes\n` +
`• GitLab MR reviewer on Vertex AI with scored LLM evaluation\n` +
`• GenAI product search with embeddings and vector databases (Milvus, Chroma, Qdrant)\n\n` +
`He works daily with Cursor and Claude Code.`
),
projects: () => (
`The work that usually matters most in conversation:\n\n` +
`• **On-Call / Data Reliability Agent** at Red Hat — MCP, LangChain, Langfuse; MTTR ~38 min → ~4 min\n` +
`• **GitOps data mesh + catalog** at Red Hat — 150+ data products, contracts, Atlan adoption 30 → 730 users\n` +
`• **Conversational AI platform** at Tata Neu — voice and text, 120M+ users, 500M events/day\n` +
`• **Voice-call analysis** — STT, audio ML, custom voice models, 12 Indic languages\n` +
`• **GenAI product search** at Tata Neu — Azure OpenAI, LangChain, vector databases\n` +
`• **LLM finance platform** at Beem — 50M+ users`
),
education: () => {
const edu = knowledgeBase.education;
return (
`• **${edu[0].school}** — ${edu[0].degree}\n ${edu[0].focus}. GPA ${edu[0].gpa}. ${edu[0].period}.\n\n` +
`• **${edu[1].school}** — ${edu[1].degree}\n GPA ${edu[1].gpa}. ${edu[1].period}.\n\n` +
`He also did information-retrieval research at ISI Kolkata under Dr. Dwaipayan Roy.`
);
},
contact: () => {
const a = knowledgeBase.about;
return (
`• **Email:** ${a.email}\n` +
`• **GitHub:** github.com/${a.github}\n` +
`• **LinkedIn:** linkedin.com/in/${a.linkedin}\n` +
`• **Website:** ${a.website}\n` +
`• **Location:** ${a.location}`
);
},
location: () => 'Arijit is based in **Bangalore, India**. He is open to conversations about India-based roles and travel where it is useful.',
hiring: () => (
`Yes — he is open to the right next role.\n\n` +
`Strongest fit: Voice AI / Voice Agents, data & strategy for model quality, agent evals, and platform engineering that production agents consume.\n\n` +
`Please write to **arijitroy003@gmail.com** or LinkedIn (**sudo-kill**). A résumé can be sent on request.`
),
currentWork: () => (
`At **Red Hat**, Arijit is Senior Software Engineer and Technical Lead for Data & AI Platform Engineering.\n\n` +
`Current focus:\n` +
`• Production agents (On-Call / Data Reliability Agent) with MCP, LangChain, and Langfuse\n` +
`• Data contracts and catalog quality for 150+ data products\n` +
`• LLM evaluation on an internal GitLab MR reviewer\n` +
`• GitOps / OpenShift data plane that those agents consume\n\n` +
`He was the first hire on the team and leads five engineers.`
),
achievements: () => (
`Selected results, all from shipped systems:\n\n` +
`• Investigation MTTR **~38 min → ~4 min** via the On-Call / Data Reliability Agent\n` +
`• **150+** data products; catalog adoption **~30 → 730** users\n` +
`• **1000+** lead engineer hours saved with Release Assistant\n` +
`• **$200k+** platform cost reduction at Red Hat\n` +
`• Conversational AI for **120M+** users and **500M** events/day at Tata Neu\n` +
`• Lakehouse migration: **75%** lower latency, **80%** lower cost\n` +
`• Voice-call quality loop across **12 Indic languages**\n` +
`• Beem: **50M+** users; Databricks funding support ($16k + $24k committed)`
),
hobbies: () => (
`Outside work he plays chess, writes technical notes, and spends time on Linux tooling. The professional through-line is the same: careful systems, clear writing, and iterative improvement.`
),
snowflake: () => (
`At Red Hat he used **Snowflake** as part of a GitOps data mesh on OpenShift, with dbt, contracts, and catalog quality gates. The platform migration away from legacy Redshift/Starburst delivered **$200k+** in cost reduction. For voice-agent work, the more relevant layer is the governed metadata and evals those warehouses now feed.`
),
databricks: () => (
`**Databricks** and **Delta Lake** show up in two places: the Tata Neu lakehouse migration (**75%** faster, **80%** cheaper) and Beem’s LLM finance platform plus 500 GB/day clickstream. He also supported a Databricks funding conversation at Beem.`
),
dbt: () => (
`**dbt** is part of the Red Hat GitOps data mesh: tested SQL, documentation, lineage, and contracts so producer changes fail at the boundary instead of in production.`
),
kubernetes: () => (
`He runs data and agent workloads on **Kubernetes** and **OpenShift** at Red Hat — GitOps deployments, CI, and the Python/Golang services that agents and evals consume.`
),
airflow: () => (
`**Airflow** orchestrates dbt, Fivetran, and custom jobs in the GitOps data mesh. The operational bar is the same as for agents: known schedules, visible failures, and quality checks before data reaches consumers.`
),
langchain: () => (
`**LangChain** is in production in two systems: GenAI product search at Tata Neu, and the On-Call / Data Reliability Agent at Red Hat (with MCP and Langfuse). He treats it as an orchestration layer with evals, not a prototype kit.`
),
mcp: () => (
`At Red Hat he builds **MCP** servers so agents can query catalogs, lineage, and operational tools under guardrails. The On-Call / Data Reliability Agent is the main example: tool-calling against governed metadata, with Langfuse traces.`
),
vectordb: () => (
`At Tata Neu he used **Milvus**, **Chroma**, and related stores for GenAI product search — embeddings plus retrieval for 120M+ users. Vector search is one piece of that quality loop, alongside intent and escalation metrics.`
),
gitops: () => (
`The Red Hat data plane is **GitOps** on OpenShift: infrastructure, pipelines, and product definitions live in Git. Release Assistant and CI are how that scales to 150+ data products without deploy-by-chat.`
),
dataops: () => (
`His DataOps practice is CI/CD for transformations, data contracts, catalog completeness, and pipeline observability — so quality issues surface before a stakeholder meeting, not during one.`
),
leadership: () => (
`Leadership experience:\n\n` +
`• First hire of Red Hat Data & AI Platform; leads **5** engineers\n` +
`• Led **6** engineers at Tata Neu on 15+ customer-facing pipelines\n` +
`• Founding engineer at Gnosis Lab\n` +
`• Enablement workshops and Agentic SDLC practices for internal teams\n\n` +
`The pattern is high ownership: 0-to-1, then quality bars, then scale.`
),
startup: () => (
`**Gnosis Lab** (NASSCOM 10K) was the founding-engineer chapter: AWS backends, an AI marketing bot, and an LMS with 50+ APIs. That same first-hire ownership shows up later as the first Data & AI Platform hire at Red Hat.`
),
industries: () => (
`• **Enterprise / open source** — Red Hat, data platforms and production agents\n` +
`• **Fintech** — Beem, personal finance for 50M+ users\n` +
`• **E-commerce / retail** — Tata Neu conversational AI and product search\n` +
`• **SaaS** — Gnosis Lab, 0-to-1 products\n\n` +
`The repeating problem is quality under scale: voice, events, and metadata that agents can trust.`
),
research: () => (
`At **ISI Kolkata** (Information Retrieval Lab, Dr. Dwaipayan Roy) he built an offline document indexer — early search/context work over unstructured corpora. Formal degrees: MCA from Jadavpur University (GPA 8.81) and B.Sc. Computer Science from Ramakrishna Mission Vidyamandira (GPA 8.49).`
),
whyData: () => (
`He works in data and voice AI because production conversations and pipelines have measurable failure modes. The useful loop is: observe a break, define a quality bar, collect or constrain the data, then measure again. That is the same loop behind Tata Neu’s voice analytics and Red Hat’s agent evals.`
),
aboutBot: () => (
`This site has two assistants:\n\n` +
`• **This text chat** — answers from a local knowledge base, with an optional in-browser LLM\n` +
`• **The voice widget** — a LiveKit agent that can speak with you\n\n` +
`Neither records conversations for later review. I only use the facts on this page.`
),
blog: () => (
`The **blog** tab has technical writing on data mesh, MCP, vector databases, and lakehouse work. Open-source and side projects include **usernaut** (Kubernetes operator), **gitlab-mr-reviewer**, and **genai-toolbox**.`
),
resume: () => (
`Please email **arijitroy003@gmail.com** or use LinkedIn (**sudo-kill**) and a current résumé will be sent. You can also review experience, projects, and contact details on this site.`
),
name: () => (
`This assistant represents **Arijit Kumar Roy**, a Data & AI platform lead in Bangalore with **${knowledgeBase.about.experience}** of experience.\n\n` +
`Current role: ${knowledgeBase.about.role} at Red Hat. Earlier: Beem, Tata Digital / Tata Neu, and founding engineer at Gnosis Lab.`
),
joke: [
'I can keep this professional — but if you would like a light one: why did the eval suite join the call? To score the conversation before anyone else did.',
'A short one: the pipeline did not fail. It raised a quality signal. That is the polite version.'
],
thankyou: [
'You are welcome. I can continue with experience, projects, or contact details whenever you are ready.',
'Glad that helped. What else would you like to know?'
],
fallback: [
'I may not have that detail. I can speak to experience, skills, projects, voice AI work, or how to get in touch.',
'That is outside this knowledge base. Try experience, Red Hat, Tata Neu, projects, or contact.'
]
};
function detectIntent(input) {
const lower = input.toLowerCase();
if (/^(hi|hello|hey|greetings|howdy|yo)\b/i.test(lower)) return 'greeting';
if (/thank|thanks|thx|appreciated/i.test(lower)) return 'thankyou';
if (/joke|funny|humor|laugh|make me laugh|tell me something funny/i.test(lower)) return 'joke';
if (/who\s*(are\s*you|is\s*(this|arijit))|your\s*name|introduce/i.test(lower)) return 'name';
if (/hiring|hire|job\s*opportunit|open\s*to|available|looking\s*for\s*(work|job|role)|recruit|opportunit|position/i.test(lower)) return 'hiring';
if (/current(ly)?|now|these\s*days|working\s*on|doing\s*now|present/i.test(lower)) return 'currentWork';
if (/red\s*hat/i.test(lower)) return 'redhat';
if (/tata|neu\b/i.test(lower)) return 'tata';
if (/beem/i.test(lower)) return 'beem';
if (/gnosis/i.test(lower)) return 'gnosis';
if (/\bsnowflake\b/i.test(lower)) return 'snowflake';
if (/\bdatabricks\b|delta\s*lake|pyspark/i.test(lower)) return 'databricks';
if (/\bdbt\b|data\s*build\s*tool/i.test(lower)) return 'dbt';
if (/kubernetes|k8s|\bkube\b/i.test(lower)) return 'kubernetes';
if (/airflow|orchestrat/i.test(lower)) return 'airflow';
if (/langchain/i.test(lower)) return 'langchain';
if (/\bmcp\b|model\s*context\s*protocol/i.test(lower)) return 'mcp';
if (/vector\s*(db|database)|milvus|chroma|qdrant|embedding/i.test(lower)) return 'vectordb';
if (/gitops/i.test(lower)) return 'gitops';
if (/dataops/i.test(lower)) return 'dataops';
if (/achievement|accomplish|impact|metric|numbers|savings|result/i.test(lower)) return 'achievements';
if (/\bblog\b|article|medium|writing|side\s*project|repos?\b/i.test(lower)) return 'blog';
if (/hobbi|interest|free\s*time|fun|outside\s*work|chess|linux/i.test(lower)) return 'hobbies';
if (/lead|leader|team|manage|mentor/i.test(lower)) return 'leadership';
if (/startup|founding|nasscom|entrepreneur/i.test(lower)) return 'startup';
if (/industr|sector|domain|fintech|ecommerce|e-commerce|retail|saas/i.test(lower)) return 'industries';
if (/research|isi|indian\s*statistical|academic|paper|publication/i.test(lower)) return 'research';
if (/why\s*(data|ai|this\s*field|engineer)|passion|motivation|interested/i.test(lower)) return 'whyData';
if (/this\s*(bot|chatbot|ai)|how\s*(do|does)\s*(this|you)\s*work|built\s*this/i.test(lower)) return 'aboutBot';
if (/resume|cv\b|curriculum/i.test(lower)) return 'resume';
if (/voice|speech|stt|indic|langfuse|eval/i.test(lower)) return 'ai';
if (/experience|work\s*history|career|roles?|companies|background/i.test(lower)) return 'experience';
if (/skills?|tech\s*stack|technologies|proficient|languages?|programming|expertise/i.test(lower)) return 'skills';
if (/\bai\b|llm|ml\b|machine\s*learning|openai|gpt|gemini|claude|mistral|genai|generative/i.test(lower)) return 'ai';
if (/projects?|built|portfolio|showcase/i.test(lower)) return 'projects';
if (/education|degree|university|college|study|school|gpa|masters|bachelor/i.test(lower)) return 'education';
if (/contact|email|linkedin|github|reach|connect|social/i.test(lower)) return 'contact';
if (/location|where|based|city|live|country|india|bangalore/i.test(lower)) return 'location';
if (/python|golang|\bgo\b|sql|javascript|typescript|scala|terraform|docker/i.test(lower)) return 'skills';
if (/aws|azure|gcp|cloud/i.test(lower)) return 'skills';
if (/kafka|fivetran|atlan|mongodb|dynamodb/i.test(lower)) return 'skills';
return 'fallback';
}
function getResponse(intent) {
const resp = responses[intent];
if (typeof resp === 'function') return resp();
if (Array.isArray(resp)) return resp[Math.floor(Math.random() * resp.length)];
return responses.fallback[0];
}