{"schema_version":"onlylabs.public_signal.v1","title":"Amazon (Nova) Writing: When LLM judges agree, should we believe them?","description":"Amazon (Nova) writing signal with public source context, captured evidence pages, related signals, and data-business radar classification.","url":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88","json_url":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88/signal.json","generated_at":"2026-08-28T17:46:56.727Z","evidence_latest_fetched_at":"2026-08-26T20:02:18.091335+00:00","signal_first_seen_at":"2026-08-26T20:00:48.481585+00:00","org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab","category_label":"Frontier lab","dossier_url":"https://onlylabs.fyi/labs/amazon","dossier_json_url":"https://onlylabs.fyi/labs/amazon/dossier.json"},"related_urls":{"signal":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88","signal_json":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88/signal.json","source":"https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them","lab_dossier":"https://onlylabs.fyi/labs/amazon","lab_dossier_json":"https://onlylabs.fyi/labs/amazon/dossier.json","analysis":"https://onlylabs.fyi/analysis/amazon","analysis_json":"https://onlylabs.fyi/analysis/amazon/analysis.json","analysis_evidence_json":"https://onlylabs.fyi/analysis/amazon/evidence.json","category":"https://onlylabs.fyi/frontier","category_json":"https://onlylabs.fyi/frontier.json","category_feed":"https://onlylabs.fyi/frontier/feed.xml","category_signals_json":"https://onlylabs.fyi/signals.json","topic":"https://onlylabs.fyi/topics/talking","topic_signals_json":"https://onlylabs.fyi/topics/talking/signals.json","topic_feed":"https://onlylabs.fyi/topics/talking/feed.xml","data_business":null},"answer_pack":{"answer":"Amazon (Nova) published When LLM judges agree, should we believe them?. This talking signal gives public context for research themes, product direction, policy, or launch framing. High-signal details: Substantive research post from Amazon on LLM judge reliability. · When LLM judges agree, should we believe them? - Amazon Science Close Close Social bluesky threads twitter instagram youtube facebook linkedin github rss Menu Research.... onlylabs links this event to 1 captured evidence page and 6 related writing signals.","signal_desk":"talking","source_context":{"source_url":"https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them","source_host":"amazon.science","occurred_at":"2026-08-26T17:10:40+00:00","first_seen_at":"2026-08-26T20:00:48.481585+00:00","date_source":"rss.item_date","context":null},"context_markers":[{"label":"Lab","value":"Amazon (Nova)","source":"signal"},{"label":"Signal desk","value":"talking","source":"signal"},{"label":"Source host","value":"amazon.science","source":"source"},{"label":"Notability","value":"Substantive research post from Amazon on LLM judge reliability.","source":"signal"},{"label":"Watch term","value":"Eval methodology","source":"evidence"},{"label":"Watch term","value":"Data pipeline","source":"evidence"},{"label":"Watch term","value":"Infrastructure","source":"evidence"},{"label":"Watch term","value":"Safety and alignment","source":"evidence"}],"evidence_coverage":{"target_pages":1,"captured_pages":1,"readable_pages":1,"capture_methods":["plain"],"missing_page_urls":[],"failed_page_urls":[],"blocked_page_urls":[],"page_urls":["https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them"],"related_signals":6,"has_source_url":true,"latest_page_fetched_at":"2026-08-26T20:02:18.091335+00:00"},"data_business":{"matches":false,"lanes":[],"matched_terms":[],"score":null,"reason":null},"agent_handoff":{"signal_json":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88/signal.json","dossier_json":"https://onlylabs.fyi/labs/amazon/dossier.json","analysis_json":"https://onlylabs.fyi/analysis/amazon/analysis.json","analysis_evidence_json":"https://onlylabs.fyi/analysis/amazon/evidence.json","topic_signals_json":"https://onlylabs.fyi/topics/talking/signals.json","topic_feed":"https://onlylabs.fyi/topics/talking/feed.xml","category_signals_json":"https://onlylabs.fyi/signals.json","data_radar_json":null,"opportunities_json":null},"analysis_playbook":{"objective":"Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.","evidence_focus":["post title","source URL","captured page text","HN traction","linked model or paper references","publication date"],"extraction_questions":["Which themes are labs choosing to explain publicly?","Which posts are attracting outside discussion?","Which writing reframes a recent release, model, hiring wave, or policy stance?","Which posts mention data, evals, infrastructure, safety, or deployment workflows?"],"signal_questions":["What public theme, launch framing, or research direction does this writing signal expose?","Which themes are labs choosing to explain publicly?","Which posts are attracting outside discussion?","Do the 6 related writing signals show a repeated pattern?"],"output_fields":["org","theme","public_framing","traction","data_business_lane","evidence_url"],"data_business_relevance":"Public writing supplies the narrative layer over raw signals and helps identify which frontier-lab priorities are becoming externally legible.","required_sources":[{"label":"signal_json","url":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88/signal.json","required":true},{"label":"source","url":"https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them","required":true},{"label":"dossier_json","url":"https://onlylabs.fyi/labs/amazon/dossier.json","required":true},{"label":"analysis_evidence_json","url":"https://onlylabs.fyi/analysis/amazon/evidence.json","required":true},{"label":"topic_signals_json","url":"https://onlylabs.fyi/topics/talking/signals.json","required":false},{"label":"data_radar_json","url":null,"required":false}],"expected_output":["one-paragraph source-grounded interpretation","category-specific implication","confidence and missing evidence","recommended next source to inspect"],"prompt_seed":"Using only the linked onlylabs JSON, captured source context, and cited evidence, analyze Amazon (Nova)'s writing signal \"When LLM judges agree, should we believe them?\" for frontier lab strategy."},"semantic_triples":[{"subject":"Amazon (Nova)","predicate":"published","object":"When LLM judges agree, should we believe them?","text":"Amazon (Nova) published When LLM judges agree, should we believe them?."},{"subject":"When LLM judges agree, should we believe them?","predicate":"is classified as","object":"writing signal","text":"When LLM judges agree, should we believe them? is classified as writing signal."},{"subject":"When LLM judges agree, should we believe them?","predicate":"belongs to","object":"talking desk","text":"When LLM judges agree, should we believe them? belongs to talking desk."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has evidence coverage","object":"1 captured evidence page","text":"When LLM judges agree, should we believe them? has evidence coverage 1 captured evidence page."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has captured page count","object":"1","text":"When LLM judges agree, should we believe them? has captured page count 1."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has readable page count","object":"1","text":"When LLM judges agree, should we believe them? has readable page count 1."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has related signal count","object":"6","text":"When LLM judges agree, should we believe them? has related signal count 6."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has analysis playbook objective","object":"Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.","text":"When LLM judges agree, should we believe them? has analysis playbook objective Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has source host","object":"amazon.science","text":"When LLM judges agree, should we believe them? has source host amazon.science."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has lab","object":"Amazon (Nova)","text":"When LLM judges agree, should we believe them? has lab Amazon (Nova)."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has signal desk","object":"talking","text":"When LLM judges agree, should we believe them? has signal desk talking."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has source host","object":"amazon.science","text":"When LLM judges agree, should we believe them? has source host amazon.science."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has notability","object":"Substantive research post from Amazon on LLM judge reliability.","text":"When LLM judges agree, should we believe them? has notability Substantive research post from Amazon on LLM judge reliability.."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has watch term","object":"Eval methodology","text":"When LLM judges agree, should we believe them? has watch term Eval methodology."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has watch term","object":"Data pipeline","text":"When LLM judges agree, should we believe them? has watch term Data pipeline."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has watch term","object":"Infrastructure","text":"When LLM judges agree, should we believe them? has watch term Infrastructure."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has watch term","object":"Safety and alignment","text":"When LLM judges agree, should we believe them? has watch term Safety and alignment."}]},"intelligence":{"signal_desk":"talking","answer":"Amazon (Nova) published When LLM judges agree, should we believe them?. This talking signal gives public context for research themes, product direction, policy, or launch framing. High-signal details: Substantive research post from Amazon on LLM judge reliability. · When LLM judges agree, should we believe them? - Amazon Science Close Close Social bluesky threads twitter instagram youtube facebook linkedin github rss Menu Research.... onlylabs links this event to 1 captured evidence page and 6 related writing signals.","semantic_triples":[{"subject":"Amazon (Nova)","predicate":"published","object":"When LLM judges agree, should we believe them?","text":"Amazon (Nova) published When LLM judges agree, should we believe them?."},{"subject":"When LLM judges agree, should we believe them?","predicate":"is classified as","object":"writing signal","text":"When LLM judges agree, should we believe them? is classified as writing signal."},{"subject":"When LLM judges agree, should we believe them?","predicate":"belongs to","object":"talking desk","text":"When LLM judges agree, should we believe them? belongs to talking desk."},{"subject":"When LLM judges agree, should we believe them?","predicate":"has evidence coverage","object":"1 captured evidence page","text":"When LLM judges agree, should we believe them? has evidence coverage 1 captured evidence page."}]},"signal":{"id":"8f86ca40-02b4-4150-b66c-7156e8da7a88","url":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88","json_url":"https://onlylabs.fyi/signals/8f86ca40-02b4-4150-b66c-7156e8da7a88/signal.json","source_url":"https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them","title":"When LLM judges agree, should we believe them?","summary":"Amazon (Nova) published a writing signal. onlylabs watches public writing for research themes, product direction, and model-launch context.","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-26T17:10:40+00:00","first_seen_at":"2026-08-26T20:00:48.481585+00:00","date_source":"rss.item_date","evidence_coverage":{"target_pages":1,"captured_pages":1,"readable_pages":1,"capture_methods":["plain"],"missing_page_urls":[],"failed_page_urls":[],"blocked_page_urls":[],"page_urls":["https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them"]},"facets":{},"traction":{"github_stars":null,"hn_points":null,"hn_comments":null,"hn_story_id":null,"hf_downloads":null,"hf_likes":null},"data_radar":null},"primary_evidence_page":{"is_primary":true,"source_match":true,"url":"https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them","final_url":"https://www.amazon.science/blog/when-llm-judges-agree-should-we-believe-them","title":"When LLM judges agree, should we believe them?","http_status":200,"content_type":"text/html;charset=UTF-8","capture_method":"plain","fetched_at":"2026-08-26T20:02:18.091335+00:00","bytes":321226,"raw_path":"5728c436a7598b96a221958e9aa58fbfcffa603952368f6713c396d941f2a1cc.html","content_hash":"b1deea6b6588ae0c28609190aa0e4c00ab84230f6422b35ae514252603a9736a","excerpt_chars":1200,"truncated":true,"excerpt":"When LLM judges agree, should we believe them? - Amazon Science Close Close Social bluesky threads twitter instagram youtube facebook linkedin github rss Menu Research Research areas Automated reasoning Cloud and systems Computer vision Conversational AI Economics Information and knowledge management Machine learning Operations research and optimization Quantum technologies Robotics Search and information retrieval Security, privacy, and abuse prevention Sustainability Our scientific contributions Publications Research from our scientists and collaborators. Conferences Our experts present and discuss cutting-edge research at scientific meetings globally. Research areas Automated reasoning Cloud and systems Computer vision Conversational AI Economics Information and knowledge management Machine learning Operations research and optimization Quantum technologies Robotics Search and information retrieval Security, privacy, and abuse prevention Sustainability Our scientific contributions Publications Research from our scientists and collaborators. Conferences Our experts present and discuss cutting-edge research at scientific meetings globally. News & blog The latest from Amazon..."},"evidence_pages":[],"related_signals":[{"id":"ad1dccae-833e-40b3-a25b-9261616f768c","url":"https://onlylabs.fyi/signals/ad1dccae-833e-40b3-a25b-9261616f768c","source_url":"https://www.amazon.science/blog/sop-bench-a-new-benchmark-for-evaluating-ai-agents-on-real-business-procedures","title":"SOP-Bench: A new benchmark for evaluating AI agents on real business procedures","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-21T15:57:17+00:00","first_seen_at":"2026-08-21T20:01:39.559676+00:00","date_source":"rss.item_date"},{"id":"4165aa6a-a827-4934-a303-9c3c22272b1d","url":"https://onlylabs.fyi/signals/4165aa6a-a827-4934-a303-9c3c22272b1d","source_url":"https://www.amazon.science/blog/a-decade-of-mathematical-certainty-reflections-on-the-automated-reasoning-group","title":"A decade of mathematical certainty: Reflections on the Automated Reasoning Group","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-11T16:22:19+00:00","first_seen_at":"2026-08-11T20:01:47.870265+00:00","date_source":"rss.item_date"},{"id":"29fdc7c2-b615-4c06-bfe2-069712889f41","url":"https://onlylabs.fyi/signals/29fdc7c2-b615-4c06-bfe2-069712889f41","source_url":"https://www.amazon.science/news/aws-trainium-frontier-competition-co-design-models-and-kernels-on-purpose-built-ai-chips","title":"AWS Trainium Frontier competition: Co-design models and kernels on purpose-built AI chips","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-10T20:23:04+00:00","first_seen_at":"2026-08-11T00:00:48.527129+00:00","date_source":"rss.item_date"},{"id":"3cb0596a-41dc-4fd2-9a38-e073a388cffa","url":"https://onlylabs.fyi/signals/3cb0596a-41dc-4fd2-9a38-e073a388cffa","source_url":"https://www.amazon.science/research-awards/latest-news/34-amazon-research-awards-build-on-trainium-recipients-announced","title":"34 Amazon Research Awards Build on Trainium recipients announced","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-08-05T15:00:00+00:00","first_seen_at":"2026-08-05T16:01:46.004825+00:00","date_source":"rss.item_date"},{"id":"ff8af5b2-8b53-479e-8d69-f20f00c837b3","url":"https://onlylabs.fyi/signals/ff8af5b2-8b53-479e-8d69-f20f00c837b3","source_url":"https://www.amazon.science/blog/how-controllers-from-industrial-machinery-can-coordinate-multitask-machine-learning","title":"How controllers from industrial machinery can coordinate multitask machine learning","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-07-30T17:26:47+00:00","first_seen_at":"2026-07-30T20:00:48.841742+00:00","date_source":"rss.item_date"},{"id":"b0590e85-a0ba-4718-bdab-f296ac1f8bc0","url":"https://onlylabs.fyi/signals/b0590e85-a0ba-4718-bdab-f296ac1f8bc0","source_url":"https://www.amazon.science/blog/a-new-benchmark-for-evaluating-patient-facing-health-ai-agents","title":"A new benchmark for evaluating patient-facing health AI agents","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"amazon","name":"Amazon (Nova)","category":"frontier-lab"},"occurred_at":"2026-07-29T15:16:52+00:00","first_seen_at":"2026-07-29T16:00:49.065318+00:00","date_source":"rss.item_date"}]}