{"schema_version":"onlylabs.public_signal.v1","title":"Amazon (Nova) Writing: SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","description":"Amazon (Nova) writing signal with public source context, captured evidence pages, related signals, and data-business radar classification.","url":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c","json_url":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c/signal.json","generated_at":"2026-08-28T17:48:15.689Z","evidence_latest_fetched_at":"2026-08-21T20:02:52.437077+00:00","signal_first_seen_at":"2026-08-21T20:01:39.559676+00:00","org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab","category_label":"Frontier lab","dossier_url":"https://onlylabs.fyi/labs/amazon","dossier_json_url":"https://onlylabs.fyi/labs/amazon/dossier.json"},"related_urls":{"signal":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c","signal_json":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c/signal.json","source":"https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures","lab_dossier":"https://onlylabs.fyi/labs/amazon","lab_dossier_json":"https://onlylabs.fyi/labs/amazon/dossier.json","analysis":"https://onlylabs.fyi/analysis/amazon","analysis_json":"https://onlylabs.fyi/analysis/amazon/analysis.json","analysis_evidence_json":"https://onlylabs.fyi/analysis/amazon/evidence.json","category":"https://onlylabs.fyi/frontier","category_json":"https://onlylabs.fyi/frontier.json","category_feed":"https://onlylabs.fyi/frontier/feed.xml","category_signals_json":"https://onlylabs.fyi/signals.json","topic":"https://onlylabs.fyi/topics/talking","topic_signals_json":"https://onlylabs.fyi/topics/talking/signals.json","topic_feed":"https://onlylabs.fyi/topics/talking/feed.xml","data_business":{"radar":"https://onlylabs.fyi/data-radar","radar_json":"https://onlylabs.fyi/data-radar.json","opportunities":"https://onlylabs.fyi/opportunities","opportunities_json":"https://onlylabs.fyi/opportunities.json","lanes":[{"key":"evals","label":"Evals and quality","url":"https://onlylabs.fyi/data-radar/evals","json_url":"https://onlylabs.fyi/data-radar/evals/signals.json"}]}},"answer_pack":{"answer":"Amazon (Nova) published SOP-Bench: A new benchmark for evaluating AI agents on real business procedures. This talking signal gives public context for research themes, product direction, policy, or launch framing. High-signal details: New benchmark from Amazon for evaluating AI agents · SOP-Bench: A new benchmark for evaluating AI agents on real business procedures - Amazon Science Close Close Social bluesky threads twitter instagram youtube facebook.... onlylabs links this event to 1 captured evidence page and 6 related writing signals. It also maps to Evals and quality in the data-business radar.","signal_desk":"talking","source_context":{"source_url":"https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures","source_host":"amazon.science","occurred_at":"2026-08-21T15:57:17+00:00","first_seen_at":"2026-08-21T20:01:39.559676+00:00","date_source":"rss.item_date","context":null},"context_markers":[{"label":"Lab","value":"Amazon (Nova)","source":"signal"},{"label":"Signal desk","value":"talking","source":"signal"},{"label":"Source host","value":"amazon.science","source":"source"},{"label":"Notability","value":"New benchmark from Amazon for evaluating AI agents","source":"signal"},{"label":"Radar lane","value":"Evals and quality","source":"radar"},{"label":"Matched term","value":"eval","source":"radar"},{"label":"Matched term","value":"benchmark","source":"radar"},{"label":"Matched term","value":"testing","source":"radar"},{"label":"Watch term","value":"Eval methodology","source":"evidence"},{"label":"Watch term","value":"Data pipeline","source":"evidence"},{"label":"Watch term","value":"Safety and alignment","source":"evidence"},{"label":"Watch term","value":"Agents and tool use","source":"evidence"}],"evidence_coverage":{"target_pages":1,"captured_pages":1,"readable_pages":1,"capture_methods":["plain"],"missing_page_urls":[],"failed_page_urls":[],"blocked_page_urls":[],"page_urls":["https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures"],"related_signals":6,"has_source_url":true,"latest_page_fetched_at":"2026-08-21T20:02:52.437077+00:00"},"data_business":{"matches":true,"lanes":[{"key":"evals","label":"Evals and quality","url":"https://onlylabs.fyi/data-radar/evals","json_url":"https://onlylabs.fyi/data-radar/evals/signals.json"}],"matched_terms":["eval","benchmark","testing"],"score":17,"reason":"Amazon (Nova) has a writing signal matching evals and quality."},"agent_handoff":{"signal_json":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c/signal.json","dossier_json":"https://onlylabs.fyi/labs/amazon/dossier.json","analysis_json":"https://onlylabs.fyi/analysis/amazon/analysis.json","analysis_evidence_json":"https://onlylabs.fyi/analysis/amazon/evidence.json","topic_signals_json":"https://onlylabs.fyi/topics/talking/signals.json","topic_feed":"https://onlylabs.fyi/topics/talking/feed.xml","category_signals_json":"https://onlylabs.fyi/signals.json","data_radar_json":"https://onlylabs.fyi/data-radar.json","opportunities_json":"https://onlylabs.fyi/opportunities.json"},"analysis_playbook":{"objective":"Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.","evidence_focus":["post title","source URL","captured page text","HN traction","linked model or paper references","publication date"],"extraction_questions":["Which themes are labs choosing to explain publicly?","Which posts are attracting outside discussion?","Which writing reframes a recent release, model, hiring wave, or policy stance?","Which posts mention data, evals, infrastructure, safety, or deployment workflows?"],"signal_questions":["What public theme, launch framing, or research direction does this writing signal expose?","Which themes are labs choosing to explain publicly?","Which posts are attracting outside discussion?","Which data-business lane explains this signal: Evals and quality?","Do the 6 related writing signals show a repeated pattern?"],"output_fields":["org","theme","public_framing","traction","data_business_lane","evidence_url"],"data_business_relevance":"Public writing supplies the narrative layer over raw signals and helps identify which frontier-lab priorities are becoming externally legible.","required_sources":[{"label":"signal_json","url":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c/signal.json","required":true},{"label":"source","url":"https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures","required":true},{"label":"dossier_json","url":"https://onlylabs.fyi/labs/amazon/dossier.json","required":true},{"label":"analysis_evidence_json","url":"https://onlylabs.fyi/analysis/amazon/evidence.json","required":true},{"label":"topic_signals_json","url":"https://onlylabs.fyi/topics/talking/signals.json","required":false},{"label":"data_radar_json","url":"https://onlylabs.fyi/data-radar.json","required":true}],"expected_output":["one-paragraph source-grounded interpretation","data-business implication","confidence and missing evidence","recommended next source to inspect"],"prompt_seed":"Using only the linked onlylabs JSON, captured source context, and cited evidence, analyze Amazon (Nova)'s writing signal \"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures\" for frontier lab strategy and data-business implications."},"semantic_triples":[{"subject":"Amazon (Nova)","predicate":"published","object":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","text":"Amazon (Nova) published SOP-Bench: A new benchmark for evaluating AI agents on real business procedures."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"is classified as","object":"writing signal","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures is classified as writing signal."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"belongs to","object":"talking desk","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures belongs to talking desk."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has evidence coverage","object":"1 captured evidence page","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has evidence coverage 1 captured evidence page."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"matches data-business lanes","object":"Evals and quality","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures matches data-business lanes Evals and quality."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has captured page count","object":"1","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has captured page count 1."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has readable page count","object":"1","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has readable page count 1."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has related signal count","object":"6","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has related signal count 6."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has analysis playbook objective","object":"Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has analysis playbook objective Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has source host","object":"amazon.science","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has source host amazon.science."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has lab","object":"Amazon (Nova)","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has lab Amazon (Nova)."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has signal desk","object":"talking","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has signal desk talking."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has source host","object":"amazon.science","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has source host amazon.science."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has notability","object":"New benchmark from Amazon for evaluating AI agents","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has notability New benchmark from Amazon for evaluating AI agents."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has radar lane","object":"Evals and quality","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has radar lane Evals and quality."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has matched term","object":"eval","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has matched term eval."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has matched term","object":"benchmark","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has matched term benchmark."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has matched term","object":"testing","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has matched term testing."}]},"intelligence":{"signal_desk":"talking","answer":"Amazon (Nova) published SOP-Bench: A new benchmark for evaluating AI agents on real business procedures. This talking signal gives public context for research themes, product direction, policy, or launch framing. High-signal details: New benchmark from Amazon for evaluating AI agents · SOP-Bench: A new benchmark for evaluating AI agents on real business procedures - Amazon Science Close Close Social bluesky threads twitter instagram youtube facebook.... onlylabs links this event to 1 captured evidence page and 6 related writing signals. It also maps to Evals and quality in the data-business radar.","semantic_triples":[{"subject":"Amazon (Nova)","predicate":"published","object":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","text":"Amazon (Nova) published SOP-Bench: A new benchmark for evaluating AI agents on real business procedures."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"is classified as","object":"writing signal","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures is classified as writing signal."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"belongs to","object":"talking desk","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures belongs to talking desk."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"has evidence coverage","object":"1 captured evidence page","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures has evidence coverage 1 captured evidence page."},{"subject":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","predicate":"matches data-business lanes","object":"Evals and quality","text":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures matches data-business lanes Evals and quality."}]},"signal":{"id":"ad1dccae-833e-40b3-a25b-9261616f768c","url":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c","json_url":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c/signal.json","source_url":"https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures","title":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","summary":"Amazon (Nova) published a writing signal. onlylabs watches public writing for research themes, product direction, and model-launch context.","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-21T15:57:17+00:00","first_seen_at":"2026-08-21T20:01:39.559676+00:00","date_source":"rss.item_date","evidence_coverage":{"target_pages":1,"captured_pages":1,"readable_pages":1,"capture_methods":["plain"],"missing_page_urls":[],"failed_page_urls":[],"blocked_page_urls":[],"page_urls":["https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures"]},"facets":{},"traction":{"github_stars":null,"hn_points":null,"hn_comments":null,"hn_story_id":null,"hf_downloads":null,"hf_likes":null},"data_radar":{"lanes":[{"key":"evals","label":"Evals and quality","url":"https://onlylabs.fyi/data-radar/evals"}],"score":17,"matched_terms":["eval","benchmark","testing"],"reason":"Amazon (Nova) has a writing signal matching evals and quality."}},"primary_evidence_page":{"is_primary":true,"source_match":true,"url":"https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures","final_url":"https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures","title":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","http_status":200,"content_type":"text/html;charset=UTF-8","capture_method":"plain","fetched_at":"2026-08-21T20:02:52.437077+00:00","bytes":319763,"raw_path":"f50a7a25c3b647ba794dfd893947c78912247152db933199be3e1d75fa79f663.html","content_hash":"1b43f09628dda61b1a8d3ae8fa67acfe69b5899bb77678c71d04461575110752","excerpt_chars":1200,"truncated":true,"excerpt":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures - Amazon Science Close Close Social bluesky threads twitter instagram youtube facebook linkedin github rss Menu Research Research areas Automated reasoning Cloud and systems Computer vision Conversational AI Economics Information and knowledge management Machine learning Operations research and optimization Quantum technologies Robotics Search and information retrieval Security, privacy, and abuse prevention Sustainability Our scientific contributions Publications Research from our scientists and collaborators. Conferences Our experts present and discuss cutting-edge research at scientific meetings globally. Research areas Automated reasoning Cloud and systems Computer vision Conversational AI Economics Information and knowledge management Machine learning Operations research and optimization Quantum technologies Robotics Search and information retrieval Security, privacy, and abuse prevention Sustainability Our scientific contributions Publications Research from our scientists and collaborators. Conferences Our experts present and discuss cutting-edge research at scientific meetings globally. News &..."},"evidence_pages":[],"related_signals":[{"id":"8f86ca40-02b4-4150-b66c-7156e8da7a88","url":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88","source_url":"https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them","title":"When LLM judges agree, should we believe them?","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-26T17:10:40+00:00","first_seen_at":"2026-08-26T20:00:48.481585+00:00","date_source":"rss.item_date"},{"id":"4165aa6a-a827-4934-a303-9c3c22272b1d","url":"https://onlylabs.fyi/signals/4165aa6a-a827-4934-a303-9c3c22272b1d","source_url":"https://www.amazon.science/blog/a-decade-of-mathematical-certainty-reflections-on-the-automated-reasoning-group","title":"A decade of mathematical certainty: Reflections on the Automated Reasoning Group","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-11T16:22:19+00:00","first_seen_at":"2026-08-11T20:01:47.870265+00:00","date_source":"rss.item_date"},{"id":"29fdc7c2-b615-4c06-bfe2-069712889f41","url":"https://onlylabs.fyi/signals/29fdc7c2-b615-4c06-bfe2-069712889f41","source_url":"https://www.amazon.science/news/aws-trainium-frontier-competition-co-design-models-and-kernels-on-purpose-built-ai-chips","title":"AWS Trainium Frontier competition: Co-design models and kernels on purpose-built AI chips","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-10T20:23:04+00:00","first_seen_at":"2026-08-11T00:00:48.527129+00:00","date_source":"rss.item_date"},{"id":"3cb0596a-41dc-4fd2-9a38-e073a388cffa","url":"https://onlylabs.fyi/signals/3cb0596a-41dc-4fd2-9a38-e073a388cffa","source_url":"https://www.amazon.science/research-awards/latest-news/34-amazon-research-awards-build-on-trainium-recipients-announced","title":"34 Amazon Research Awards Build on Trainium recipients announced","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-05T15:00:00+00:00","first_seen_at":"2026-08-05T16:01:46.004825+00:00","date_source":"rss.item_date"},{"id":"ff8af5b2-8b53-479e-8d69-f20f00c837b3","url":"https://onlylabs.fyi/signals/ff8af5b2-8b53-479e-8d69-f20f00c837b3","source_url":"https://www.amazon.science/blog/how-controllers-from-industrial-machinery-can-coordinate-multitask-machine-learning","title":"How controllers from industrial machinery can coordinate multitask machine learning","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-07-30T17:26:47+00:00","first_seen_at":"2026-07-30T20:00:48.841742+00:00","date_source":"rss.item_date"},{"id":"b0590e85-a0ba-4718-bdab-f296ac1f8bc0","url":"https://onlylabs.fyi/signals/b0590e85-a0ba-4718-bdab-f296ac1f8bc0","source_url":"https://www.amazon.science/blog/a-new-benchmark-for-evaluating-patient-facing-health-ai-agents","title":"A new benchmark for evaluating patient-facing health AI agents","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-07-29T15:16:52+00:00","first_seen_at":"2026-07-29T16:00:49.065318+00:00","date_source":"rss.item_date"}]}