{"id":"qwen3-support-analysis-60","title":"Qwen3 support analysis benchmark on 60 synthetic messages","description":"Three independent ReplyPilot runs measuring topic stability, risk output, generated assets, latency, token usage, and estimated inference cost.","publishedAt":"2026-07-24","model":"@cf/qwen/qwen3-30b-a3b-fp8","provider":"Cloudflare Workers AI","dataset":{"name":"ReplyPilot synthetic support messages","messageCount":60,"format":"CSV","url":"https://replypilot.slashgallery.com/samples/sample-support.csv","license":"https://creativecommons.org/licenses/by/4.0/","fields":["ticket_id","created_at","channel","status","subject","body","tags"]},"method":{"runs":3,"maxRows":100,"stages":["topic clustering","report composition"],"deterministicSafetyPass":true,"temperature":0.2,"jsonMode":true,"thinkingMode":"disabled with /no_think"},"stableResult":{"readinessScore":68,"decision":"needs validation","topicAgreement":"3 of 3 runs returned the same topic names, volumes, and risk levels","topics":[{"name":"Shipping & tracking","volume":19,"level":"yellow"},{"name":"Order edits & promos","volume":13,"level":"yellow"},{"name":"Returns, refunds & replacements","volume":10,"level":"yellow"},{"name":"Risk, legal & escalation","volume":8,"level":"red"},{"name":"Product info & fit","volume":8,"level":"green"},{"name":"Billing, invoices & subscriptions","volume":1,"level":"yellow"},{"name":"Other support questions","volume":1,"level":"yellow"}]},"runs":[{"run":1,"durationMs":11660,"totalTokens":5540,"estimatedCostUsd":0.000604,"topicCount":7,"highRiskExamples":3,"faqDrafts":0,"actionAssets":7},{"run":2,"durationMs":8521,"totalTokens":5342,"estimatedCostUsd":0.000586,"topicCount":7,"highRiskExamples":3,"faqDrafts":3,"actionAssets":10},{"run":3,"durationMs":15840,"totalTokens":6085,"estimatedCostUsd":0.000774,"topicCount":7,"highRiskExamples":8,"faqDrafts":7,"actionAssets":17}],"aggregates":{"averageDurationMs":12007,"averageTotalTokens":5656,"averageEstimatedCostUsd":0.000655,"actionAssetRange":[7,17],"faqDraftRange":[0,7],"highRiskExampleRange":[3,8]},"representativeRiskExamples":[{"ticket":"T-1006","risk":"Duplicate charge, escalated complaint","action":"Human billing review only"},{"ticket":"T-1015","risk":"Escalated complaint","action":"Human support owner review"},{"ticket":"T-1021","risk":"Security or fraud","action":"Verify identity and escalate"}],"conclusions":["Topic grouping was stable across all three runs.","Generated FAQ, risk-example, and action-asset counts were not stable enough to use without review.","The average model inference estimate was below one tenth of one cent per 60-message analysis.","The benchmark validates a technical workflow, not customer savings, production accuracy, or willingness to pay."],"limitations":["The dataset is synthetic and contains only 60 English-language support messages.","The benchmark does not include human labels for topic purity or a complete red-risk recall score.","Costs use Cloudflare model pricing available when the benchmark was published and may change.","Latency reflects three requests from one deployment and is not a service-level guarantee.","No real customer adoption, edit-rate, or time-saved evidence is included."]}