[{"data":1,"prerenderedAt":170},["ShallowReactive",2],{"case-study-en-live-product-data-pipeline":3,"case-study-related-en-live-product-data-pipeline":147},{"id":4,"title":5,"card":6,"diagrams":12,"extension":24,"eyebrow":25,"facts":26,"intro":36,"layout":37,"locale":53,"meta":54,"order":48,"results":55,"slug":64,"stem":65,"storySections":66,"techGroups":73,"__hash__":146},"caseStudies\u002Fcase-studies\u002Fen\u002Flive-product-data-pipeline.json","AWS architecture for a live product data pipeline, from nightly batch to Kafka Streams",{"title":7,"badges":8,"thumb":11},"Live product data pipeline",[9,10],"Kafka Streams","Amazon MSK","\u002Fcase-studies\u002Flive-product-data-pipeline\u002Fafter.webp",[13,17,20],{"src":14,"alt":15,"caption":16},"\u002Fcase-studies\u002Flive-product-data-pipeline\u002Fbefore.webp","Before: A nightly batch, orchestrated end to end","The old architecture, which I also maintained.",{"src":11,"alt":18,"caption":19},"After: Kafka Streams on Amazon MSK, processing live","The new architecture, built with the team.",{"src":21,"alt":22,"caption":23},"\u002Fcase-studies\u002Flive-product-data-pipeline\u002Fcore-stack.webp","Core stack","","json","Case study",[27,30,33],{"label":28,"value":29},"Industry","E-commerce",{"label":31,"value":32},"My role","Maintained the batch pipeline, built its streaming replacement with the team",{"label":34,"value":35},"Scale","Millions of products, processed continuously","Millions of products, and a data pipeline that processed them once a night, when it didn't fail.",[38,40,42,45,47,49,51],{"type":39},"header",{"type":41},"results",{"type":43,"index":44},"section",0,{"type":46,"index":44},"diagram",{"type":46,"index":48},1,{"type":50},"technology",{"type":46,"index":52},2,"en",{},[56,60],{"tag":57,"title":58,"body":59},"Reliability","A stable job that runs reliably","No more failed nightly runs or manual replays. The streaming pipeline processes each change on its own, inside private subnets with scoped security groups.",{"tag":61,"title":62,"body":63},"Revenue","More revenue from fresher data","Decisions are based on current data instead of yesterday's, so results are better and reach customers sooner.","live-product-data-pipeline","case-studies\u002Fen\u002Flive-product-data-pipeline",[67],{"heading":68,"paragraphs":69},"The story",[70,71,72],"Every product in the catalogue needs to be processed and published based on business rules. For millions of products that ran as a nightly batch job. Step Functions orchestrated SNS, Lambdas, ECS jobs, EC2 and DynamoDB, written in Python and TypeScript and provisioned with Terraform.","I maintained that pipeline, and it failed often. It was a long chain of steps in one fixed window, so one broken step left the data stale until the next night.","With the team I rebuilt it as a Kafka Streams application on Amazon MSK, running on ECS inside private subnets. Changes now flow through Kafka and are processed as they happen, instead of once a night.",[74,97,110],{"title":75,"items":76},"Languages and tooling",[77,81,85,89,93],{"name":78,"role":79,"icon":80},"Python","data processing","\u002Ficons\u002Ftech\u002Fpython.svg",{"name":82,"role":83,"icon":84},"TypeScript","services and tooling","\u002Ficons\u002Ftech\u002Ftypescript.svg",{"name":86,"role":87,"icon":88},"Vue","internal tooling","\u002Ficons\u002Ftech\u002Fvuejs.svg",{"name":90,"role":91,"icon":92},"Docker","containers","\u002Ficons\u002Ftech\u002Fdocker.svg",{"name":94,"role":95,"icon":96},"Terraform","infrastructure as code","\u002Ficons\u002Ftech\u002Fterraform.svg",{"title":98,"items":99},"Streaming",[100,103,106],{"name":9,"role":101,"icon":102},"live processing app","\u002Ficons\u002Ftech\u002Fapachekafka.svg",{"name":10,"role":104,"icon":105},"managed Kafka cluster","\u002Ficons\u002Faws\u002FAmazon-Managed-Streaming-for-Apache-Kafka.svg",{"name":107,"role":108,"icon":109},"Amazon ECS","runs the streams app","\u002Ficons\u002Faws\u002FAmazon-Elastic-Container-Service.svg",{"title":111,"items":112},"Batch pipeline and platform",[113,117,123,128,134,140],{"name":114,"role":115,"icon":116},"AWS Step Functions","nightly orchestration","\u002Ficons\u002Faws\u002FAWS-Step-Functions.svg",{"name":118,"role":119,"icon":120,"icons":121},"AWS Lambda, Amazon SNS","external data intake","\u002Ficons\u002Faws\u002FAWS-Lambda.svg",[120,122],"\u002Ficons\u002Faws\u002FAmazon-SNS.svg",{"name":124,"role":125,"icon":126,"icons":127},"Amazon EC2, ECS","processing workflow","\u002Ficons\u002Faws\u002FAmazon-EC2.svg",[126,109],{"name":129,"role":130,"icon":131,"icons":132},"Amazon DynamoDB, RDS","internal product data","\u002Ficons\u002Faws\u002FAmazon-DynamoDB.svg",[131,133],"\u002Ficons\u002Faws\u002FAmazon-RDS.svg",{"name":135,"role":136,"icon":137,"icons":138},"Amazon VPC, IAM","networking, security groups, access","\u002Ficons\u002Faws\u002FAmazon-VPC.svg",[137,139],"\u002Ficons\u002Faws\u002FAWS-IAM.svg",{"name":141,"role":142,"icon":143,"icons":144},"Amazon CloudWatch, Grafana","metrics and error reports","\u002Ficons\u002Faws\u002FAmazon-CloudWatch.svg",[143,145],"\u002Ficons\u002Ftech\u002Fgrafana.svg","UHa_dDbxoOp8NAncvQh5j9lzqpVMMf2FYqodhwy1M4I",[148,156,163],{"slug":149,"card":150},"product-classification",{"title":151,"badges":152,"thumb":155},"Product classification with MLOps",[153,154],"Amazon Bedrock","PyTorch","\u002Fcase-studies\u002Fproduct-classification\u002Ftraining-pipeline.webp",{"slug":157,"card":158},"property-management-platform",{"title":159,"badges":160,"thumb":162},"Multi-tenant property platform",[161,153],"Multi-tenancy","\u002Fcase-studies\u002Fproperty-management-platform\u002Fsystem-architecture.webp",{"slug":164,"card":165},"ale-business-bookkeeping",{"title":166,"badges":167,"thumb":169},"Chat-driven bookkeeping",[168,153],"LLM harness","\u002Fcase-studies\u002Fale-business-bookkeeping\u002Fsystem-architecture.webp",1790607906229]