[{"data":1,"prerenderedAt":227},["ShallowReactive",2],{"profile":3,"work-ai-interviewer":14},{"name":4,"role":5,"tagline":6,"location":7,"contact":8},"Sankar Vema","AI Builder & Architect of Agentic Systems","I help enterprise leaders turn AI ambition into capability that actually ships.","India · open to global advisory engagements",{"email":9,"linkedin":10,"github":11,"blog":12,"twitter":13},"sankar.vema@gmail.com","https:\u002F\u002Fwww.linkedin.com\u002Fin\u002Fsankarvema\u002F","http:\u002F\u002Fsankarvema.github.io\u002F","http:\u002F\u002Fsankarvema.blogspot.com\u002F","https:\u002F\u002Ftwitter.com\u002Fsansvema",{"id":15,"title":16,"body":17,"context":198,"description":189,"domain":199,"draft":200,"extension":201,"hero":202,"impact":203,"meta":204,"navigation":205,"path":206,"role":207,"scale":208,"seo":209,"slug":210,"stack":211,"status":218,"stem":219,"tags":220,"year":225,"__hash__":226},"work\u002Fwork\u002Fai-interviewer.md","Real-Time Audio-Video Human-Interaction Agents",{"type":18,"value":19,"toc":188},"minimark",[20,25,29,33,36,59,63,96,100,167,171,174,177],[21,22,24],"h2",{"id":23},"the-mandate","The mandate",[26,27,28],"p",{},"I architected and led the build of a real-time conversational agent that had to\nhold a natural voice and video conversation, and produce a structured, comparable\nevaluation at the end of it. The hard part was never the language model. It was\nbuilding a system that stays conversational under a latency budget most LLM stacks\nquietly break.",[21,30,32],{"id":31},"the-problem","The problem",[26,34,35],{},"Recruiting teams burn the most expensive hour of every funnel on first-round\nscreenings. Most candidates do not survive that hour. The work is repetitive, the\nrubric is consistent, the conversation is bounded, in other words a textbook job\nfor an agent. But textbook hides three real problems:",[37,38,39,47,53],"ol",{},[40,41,42,46],"li",{},[43,44,45],"strong",{},"Latency."," Voice conversations break the moment turn-taking lag crosses\nroughly 700 ms. Most LLM stacks are not built for that budget.",[40,48,49,52],{},[43,50,51],{},"Structure."," Hiring teams do not need transcripts. They need a rubric-aligned,\ncomparable, structured evaluation per candidate.",[40,54,55,58],{},[43,56,57],{},"Mode."," A real interview is voice plus video plus screen and sometimes code.\nSingle-mode bots feel like phone trees.",[21,60,62],{"id":61},"the-architecture","The architecture",[64,65,66,72,78,84,90],"ul",{},[40,67,68,71],{},[43,69,70],{},"Orchestration:"," Pipecat for pipeline-style agent composition (ASR, then\nreasoning, then TTS, with interruption handling).",[40,73,74,77],{},[43,75,76],{},"Transport:"," Daily.co for WebRTC video and audio with a WebSocket fallback,\nselected over rolling our own to compress build time on a non-differentiating layer.",[40,79,80,83],{},[43,81,82],{},"Reasoning core:"," the LLM is constrained to a custom evaluation JSON schema\nper candidate and per round, so the output is queryable, comparable and\nrubric-aligned at write time, not parsed after the fact.",[40,85,86,89],{},[43,87,88],{},"Latency budget:"," a total round-trip target under one second. Optimized\nWebSocket transitions and parallelized ASR and reasoning to keep the floor low.",[40,91,92,95],{},[43,93,94],{},"Round routing:"," technical, HR and managerial flows are configuration, not\ncode, so recruiters compose new flows without engineering.",[21,97,99],{"id":98},"key-decisions-and-what-they-cost","Key decisions and what they cost",[101,102,103,119],"table",{},[104,105,106],"thead",{},[107,108,109,113,116],"tr",{},[110,111,112],"th",{},"Decision",[110,114,115],{},"Why",[110,117,118],{},"What it traded",[120,121,122,134,145,156],"tbody",{},[107,123,124,128,131],{},[125,126,127],"td",{},"Pipecat over a custom orchestrator",[125,129,130],{},"Mature interruption and barge-in semantics out of the box",[125,132,133],{},"Some flexibility on novel turn-taking patterns",[107,135,136,139,142],{},[125,137,138],{},"Daily.co over self-hosted WebRTC",[125,140,141],{},"Time to production",[125,143,144],{},"Per-minute cost above a usage threshold",[107,146,147,150,153],{},[125,148,149],{},"Structured JSON output over free text",[125,151,152],{},"Comparability across candidates",[125,154,155],{},"Some loss of qualitative texture, contained with a notes field",[107,157,158,161,164],{},[125,159,160],{},"Sub-second target",[125,162,163],{},"Conversational naturalness",[125,165,166],{},"Forced eager inference and parallelization, and real compute cost",[21,168,170],{"id":169},"outcome","Outcome",[26,172,173],{},"A screening agent that talks like a person and reports like a rubric. Recruiters\ncompose the interview flow, the system runs it, and the evaluation lands\nstructured and comparable across every candidate.",[175,176],"hr",{},[26,178,179,182,183],{},[43,180,181],{},"Related writing:"," ",[184,185,187],"a",{"href":186},"\u002Fwriting\u002Fagentic-architecture-patterns","Five patterns I keep reaching for when designing agentic systems",{"title":189,"searchDepth":190,"depth":190,"links":191},"",3,[192,194,195,196,197],{"id":23,"depth":193,"text":24},2,{"id":31,"depth":193,"text":32},{"id":61,"depth":193,"text":62},{"id":98,"depth":193,"text":99},{"id":169,"depth":193,"text":170},"Brane Enterprises","ai-products",false,"md","A voice and video AI that conducts technical, HR and managerial screenings end to end, with sub-second response and structured evaluation output.","Full voice and video screening under a one-second turn budget",{},true,"\u002Fwork\u002Fai-interviewer","Architect and Technical Lead","Production, sub-second latency budget",{"title":16,"description":189},"ai-interviewer",[212,213,214,215,216,217],"Pipecat","Daily.co","WebRTC","WebSocket","LLM","JSON Schema","shipped","work\u002Fai-interviewer",[221,222,223,224],"build","agentic","multimodal","latency","2024","GdNg-XnZz8BFTHsRWAZiEPz51_QpBpE0bAOXG4KH0ps",1790601541879]