[{"data":1,"prerenderedAt":158},["ShallowReactive",2],{"case-study-en-product-classification":3,"case-study-related-en-product-classification":135},{"id":4,"title":5,"card":6,"diagrams":12,"extension":24,"eyebrow":25,"facts":26,"intro":36,"layout":37,"locale":53,"meta":54,"order":52,"results":55,"slug":64,"stem":65,"storySections":66,"techGroups":73,"__hash__":134},"caseStudies\u002Fcase-studies\u002Fen\u002Fproduct-classification.json","MLOps on AWS: training and serving a product classification model",{"title":7,"badges":8,"thumb":11},"Product classification with MLOps",[9,10],"Amazon Bedrock","PyTorch","\u002Fcase-studies\u002Fproduct-classification\u002Ftraining-pipeline.webp",[13,16,20],{"src":11,"alt":14,"caption":15},"Training pipeline: From product data to a versioned model","Training pipeline.",{"src":17,"alt":18,"caption":19},"\u002Fcase-studies\u002Fproduct-classification\u002Finference-pipeline.webp","Inference pipeline: From raw product data to structured results","Inference pipeline.",{"src":21,"alt":22,"caption":23},"\u002Fcase-studies\u002Fproduct-classification\u002Fcore-stack.webp","Core stack","","json","Case study",[27,30,33],{"label":28,"value":29},"Industry","E-commerce",{"label":31,"value":32},"My role","Built the training and inference pipelines",{"label":34,"value":35},"Focus","MLOps, model training","A large online retailer, a catalogue of millions of products, and a model that turns raw product data into consistent, structured information.",[38,40,42,45,47,49,51],{"type":39},"header",{"type":41},"results",{"type":43,"index":44},"section",0,{"type":46,"index":44},"diagram",{"type":46,"index":48},1,{"type":50},"technology",{"type":46,"index":52},2,"en",{},[56,60],{"tag":57,"title":58,"body":59},"Traffic","More traffic through better findability","Consistently classified products are easier to find in search and navigation.",{"tag":61,"title":62,"body":63},"Revenue","More revenue from better product data","Products that are found more often are products that sell more often.","product-classification","case-studies\u002Fen\u002Fproduct-classification",[67],{"heading":68,"paragraphs":69},"The story",[70,71,72],"On a catalogue with millions of products, product data arrives incomplete and inconsistent. Doing classification by hand does not scale, so the goal was a trained model that does it automatically, based on different kinds of product data.","I built the training pipeline. Python jobs read product data from DynamoDB and RDS, prepare features, create text embeddings with Amazon Bedrock and train the model. Data and versioned model artifacts live in S3.","I also built the inference pipeline, a Python service on ECS that classifies new and changed products and writes the results back to the catalogue. It is monitored in CloudWatch and shipped through GitHub CI\u002FCD.",[74,104,123],{"title":75,"items":76},"AWS",[77,80,84,88,92,96,100],{"name":9,"role":78,"icon":79},"text embeddings for product data","\u002Ficons\u002Faws\u002FAmazon-Bedrock.svg",{"name":81,"role":82,"icon":83},"Amazon ECS","feature jobs, inference service","\u002Ficons\u002Faws\u002FAmazon-Elastic-Container-Service.svg",{"name":85,"role":86,"icon":87},"Amazon EC2","model training","\u002Ficons\u002Faws\u002FAmazon-EC2.svg",{"name":89,"role":90,"icon":91},"Amazon S3","training data, model artifacts","\u002Ficons\u002Faws\u002FAmazon-S3.svg",{"name":93,"role":94,"icon":95},"Amazon DynamoDB","product data","\u002Ficons\u002Faws\u002FAmazon-DynamoDB.svg",{"name":97,"role":98,"icon":99},"Amazon RDS","relational product data","\u002Ficons\u002Faws\u002FAmazon-RDS.svg",{"name":101,"role":102,"icon":103},"Amazon CloudWatch","metrics, logs, alarms","\u002Ficons\u002Faws\u002FAmazon-CloudWatch.svg",{"title":105,"items":106},"Machine learning",[107,111,113,117],{"name":108,"role":109,"icon":110},"Python","pipelines and service","\u002Ficons\u002Ftech\u002Fpython.svg",{"name":10,"role":86,"icon":112},"\u002Ficons\u002Ftech\u002Fpytorch.svg",{"name":114,"role":115,"icon":116},"scikit-learn","baselines and evaluation","\u002Ficons\u002Ftech\u002Fscikitlearn.svg",{"name":118,"role":119,"icon":120,"icons":121},"pandas, NumPy","feature preparation","\u002Ficons\u002Ftech\u002Fpandas.svg",[120,122],"\u002Ficons\u002Ftech\u002Fnumpy.svg",{"title":124,"items":125},"Delivery",[126,130],{"name":127,"role":128,"icon":129},"GitHub Actions","CI\u002FCD for pipelines and service","\u002Ficons\u002Ftech\u002Fgithubactions.svg",{"name":131,"role":132,"icon":133},"Docker","containers","\u002Ficons\u002Ftech\u002Fdocker.svg","9qWOTjXSdgjsmMVebisHma622g4l247CfLx1AkC3ic8",[136,143,150],{"slug":137,"card":138},"property-management-platform",{"title":139,"badges":140,"thumb":142},"Multi-tenant property platform",[141,9],"Multi-tenancy","\u002Fcase-studies\u002Fproperty-management-platform\u002Fsystem-architecture.webp",{"slug":144,"card":145},"ale-business-bookkeeping",{"title":146,"badges":147,"thumb":149},"Chat-driven bookkeeping",[148,9],"LLM harness","\u002Fcase-studies\u002Fale-business-bookkeeping\u002Fsystem-architecture.webp",{"slug":151,"card":152},"kv-cache-storage-tiering",{"title":153,"badges":154,"thumb":157},"KV cache storage tiering",[155,156],"vLLM","LMCache","\u002Fcase-studies\u002Fkv-cache-storage-tiering\u002Ftier-hierarchy.webp",1790607906241]