{"schema_version":"onlylabs.public_signal.v1","title":"OpenAI Writing: How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","description":"OpenAI writing signal with public source context, captured evidence pages, related signals, and data-business radar classification.","url":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d","json_url":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d/signal.json","generated_at":"2026-07-30T04:50:58.727Z","evidence_latest_fetched_at":"2026-07-30T00:03:11.144+00:00","signal_first_seen_at":"2026-07-30T00:01:50.282493+00:00","org":{"slug":"openai","name":"OpenAI","category":"frontier-lab","category_label":"Frontier lab","dossier_url":"https://onlylabs.fyi/labs/openai","dossier_json_url":"https://onlylabs.fyi/labs/openai/dossier.json"},"related_urls":{"signal":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d","signal_json":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d/signal.json","source":"https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores","lab_dossier":"https://onlylabs.fyi/labs/openai","lab_dossier_json":"https://onlylabs.fyi/labs/openai/dossier.json","analysis":"https://onlylabs.fyi/analysis/openai","analysis_json":"https://onlylabs.fyi/analysis/openai/analysis.json","analysis_evidence_json":"https://onlylabs.fyi/analysis/openai/evidence.json","category":"https://onlylabs.fyi/frontier","category_json":"https://onlylabs.fyi/frontier.json","category_feed":"https://onlylabs.fyi/frontier/feed.xml","category_signals_json":"https://onlylabs.fyi/signals.json","topic":"https://onlylabs.fyi/topics/talking","topic_signals_json":"https://onlylabs.fyi/topics/talking/signals.json","topic_feed":"https://onlylabs.fyi/topics/talking/feed.xml","data_business":{"radar":"https://onlylabs.fyi/data-radar","radar_json":"https://onlylabs.fyi/data-radar.json","opportunities":"https://onlylabs.fyi/opportunities","opportunities_json":"https://onlylabs.fyi/opportunities.json","lanes":[{"key":"evals","label":"Evals and quality","url":"https://onlylabs.fyi/data-radar/evals","json_url":"https://onlylabs.fyi/data-radar/evals/signals.json"}]}},"answer_pack":{"answer":"OpenAI published How enabling two settings tripled our scores on the ARC-AGI-3 benchmark. This talking signal gives public context for research themes, product direction, policy, or launch framing. High-signal details: How enabling two settings tripled our scores on the ARC-AGI-3 benchmark \\| OpenAI July 29, 2026 [Research](https://openai.com/news/research/).... onlylabs links this event to 1 captured evidence page and 6 related writing signals. It also maps to Evals and quality in the data-business radar.","signal_desk":"talking","source_context":{"source_url":"https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores","source_host":"openai.com","occurred_at":"2026-07-29T15:00:00+00:00","first_seen_at":"2026-07-30T00:01:50.282493+00:00","date_source":"rss.item_date","context":null},"context_markers":[{"label":"Lab","value":"OpenAI","source":"signal"},{"label":"Signal desk","value":"talking","source":"signal"},{"label":"Source host","value":"openai.com","source":"source"},{"label":"Radar lane","value":"Evals and quality","source":"radar"},{"label":"Matched term","value":"benchmark","source":"radar"},{"label":"Watch term","value":"Eval methodology","source":"evidence"},{"label":"Watch term","value":"Data pipeline","source":"evidence"},{"label":"Watch term","value":"Agents and tool use","source":"evidence"}],"evidence_coverage":{"target_pages":1,"captured_pages":1,"readable_pages":1,"capture_methods":["firecrawl"],"missing_page_urls":[],"failed_page_urls":[],"blocked_page_urls":[],"page_urls":["https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores"],"related_signals":6,"has_source_url":true,"latest_page_fetched_at":"2026-07-30T00:03:11.144+00:00"},"data_business":{"matches":true,"lanes":[{"key":"evals","label":"Evals and quality","url":"https://onlylabs.fyi/data-radar/evals","json_url":"https://onlylabs.fyi/data-radar/evals/signals.json"}],"matched_terms":["benchmark"],"score":14,"reason":"OpenAI has a writing signal matching evals and quality."},"agent_handoff":{"signal_json":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d/signal.json","dossier_json":"https://onlylabs.fyi/labs/openai/dossier.json","analysis_json":"https://onlylabs.fyi/analysis/openai/analysis.json","analysis_evidence_json":"https://onlylabs.fyi/analysis/openai/evidence.json","topic_signals_json":"https://onlylabs.fyi/topics/talking/signals.json","topic_feed":"https://onlylabs.fyi/topics/talking/feed.xml","category_signals_json":"https://onlylabs.fyi/signals.json","data_radar_json":"https://onlylabs.fyi/data-radar.json","opportunities_json":"https://onlylabs.fyi/opportunities.json"},"analysis_playbook":{"objective":"Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.","evidence_focus":["post title","source URL","captured page text","HN traction","linked model or paper references","publication date"],"extraction_questions":["Which themes are labs choosing to explain publicly?","Which posts are attracting outside discussion?","Which writing reframes a recent release, model, hiring wave, or policy stance?","Which posts mention data, evals, infrastructure, safety, or deployment workflows?"],"signal_questions":["What public theme, launch framing, or research direction does this writing signal expose?","Which themes are labs choosing to explain publicly?","Which posts are attracting outside discussion?","Which data-business lane explains this signal: Evals and quality?","Do the 6 related writing signals show a repeated pattern?"],"output_fields":["org","theme","public_framing","traction","data_business_lane","evidence_url"],"data_business_relevance":"Public writing supplies the narrative layer over raw signals and helps identify which frontier-lab priorities are becoming externally legible.","required_sources":[{"label":"signal_json","url":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d/signal.json","required":true},{"label":"source","url":"https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores","required":true},{"label":"dossier_json","url":"https://onlylabs.fyi/labs/openai/dossier.json","required":true},{"label":"analysis_evidence_json","url":"https://onlylabs.fyi/analysis/openai/evidence.json","required":true},{"label":"topic_signals_json","url":"https://onlylabs.fyi/topics/talking/signals.json","required":false},{"label":"data_radar_json","url":"https://onlylabs.fyi/data-radar.json","required":true}],"expected_output":["one-paragraph source-grounded interpretation","data-business implication","confidence and missing evidence","recommended next source to inspect"],"prompt_seed":"Using only the linked onlylabs JSON, captured source context, and cited evidence, analyze OpenAI's writing signal \"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark\" for frontier lab strategy and data-business implications."},"semantic_triples":[{"subject":"OpenAI","predicate":"published","object":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","text":"OpenAI published How enabling two settings tripled our scores on the ARC-AGI-3 benchmark."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"is classified as","object":"writing signal","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark is classified as writing signal."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"belongs to","object":"talking desk","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark belongs to talking desk."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has evidence coverage","object":"1 captured evidence page","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has evidence coverage 1 captured evidence page."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"matches data-business lanes","object":"Evals and quality","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark matches data-business lanes Evals and quality."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has captured page count","object":"1","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has captured page count 1."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has readable page count","object":"1","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has readable page count 1."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has related signal count","object":"6","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has related signal count 6."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has analysis playbook objective","object":"Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has analysis playbook objective Turn public writing and discussion into a readable map of research themes, product framing, policy posture, launch narratives, and market attention.."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has source host","object":"openai.com","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has source host openai.com."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has lab","object":"OpenAI","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has lab OpenAI."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has signal desk","object":"talking","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has signal desk talking."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has source host","object":"openai.com","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has source host openai.com."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has radar lane","object":"Evals and quality","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has radar lane Evals and quality."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has matched term","object":"benchmark","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has matched term benchmark."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has watch term","object":"Eval methodology","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has watch term Eval methodology."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has watch term","object":"Data pipeline","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has watch term Data pipeline."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has watch term","object":"Agents and tool use","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has watch term Agents and tool use."}]},"intelligence":{"signal_desk":"talking","answer":"OpenAI published How enabling two settings tripled our scores on the ARC-AGI-3 benchmark. This talking signal gives public context for research themes, product direction, policy, or launch framing. High-signal details: How enabling two settings tripled our scores on the ARC-AGI-3 benchmark \\| OpenAI July 29, 2026 [Research](https://openai.com/news/research/).... onlylabs links this event to 1 captured evidence page and 6 related writing signals. It also maps to Evals and quality in the data-business radar.","semantic_triples":[{"subject":"OpenAI","predicate":"published","object":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","text":"OpenAI published How enabling two settings tripled our scores on the ARC-AGI-3 benchmark."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"is classified as","object":"writing signal","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark is classified as writing signal."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"belongs to","object":"talking desk","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark belongs to talking desk."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"has evidence coverage","object":"1 captured evidence page","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark has evidence coverage 1 captured evidence page."},{"subject":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","predicate":"matches data-business lanes","object":"Evals and quality","text":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark matches data-business lanes Evals and quality."}]},"signal":{"id":"a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d","url":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d","json_url":"https://onlylabs.fyi/signals/a38b1986-d7ef-4e00-a5a9-8d5267bc8e2d/signal.json","source_url":"https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores","title":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","summary":"OpenAI published a writing signal. onlylabs watches public writing for research themes, product direction, and model-launch context.","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"openai","name":"OpenAI","category":"frontier-lab"},"occurred_at":"2026-07-29T15:00:00+00:00","first_seen_at":"2026-07-30T00:01:50.282493+00:00","date_source":"rss.item_date","evidence_coverage":{"target_pages":1,"captured_pages":1,"readable_pages":1,"capture_methods":["firecrawl"],"missing_page_urls":[],"failed_page_urls":[],"blocked_page_urls":[],"page_urls":["https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores"]},"facets":{},"traction":{"github_stars":null,"hn_points":3,"hn_comments":0,"hn_story_id":"49104184","hf_downloads":null,"hf_likes":null},"data_radar":{"lanes":[{"key":"evals","label":"Evals and quality","url":"https://onlylabs.fyi/data-radar/evals"}],"score":14,"matched_terms":["benchmark"],"reason":"OpenAI has a writing signal matching evals and quality."}},"primary_evidence_page":{"is_primary":true,"source_match":true,"url":"https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores","final_url":"https://openai.com/index/how-two-settings-tripled-our-arc-agi-3-scores","title":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark","http_status":200,"content_type":"text/html","capture_method":"firecrawl","fetched_at":"2026-07-30T00:03:11.144+00:00","bytes":null,"raw_path":"265c6a0134aba9b6d5ff886a2c1062b903d9dc8f0af87700b1f7d6960b8f6865.html","content_hash":null,"excerpt_chars":1200,"truncated":true,"excerpt":"How enabling two settings tripled our scores on the ARC-AGI-3 benchmark \\| OpenAI July 29, 2026 [Research](https://openai.com/news/research/) [Publication](https://openai.com/research/index/publication/) How enabling two settings tripled our scores on the ARC-AGI-3 benchmark Loading… Share 00:00 _A sped-up video of GPT‑5.6 Sol attempting to solve puzzles in the ARC-AGI-3 benchmark, with the official harness (left) and our Responses API harness (right), which retains reasoning and enables compaction. On the leaderboard for_ [_this game_⁠(opens in a new window)](https://arcprize.org/tasks/cd82) _, no frontier model solves any level beyond the first. With our harness, GPT‑5.6 Sol solves all six._ When we first saw GPT‑5.6 Sol’s low scores on the [ARC-AGI-3⁠(opens in a new window)](https://arcprize.org/arc-agi/3) benchmark, we were puzzled. GPT‑5.6 Sol has solved longstanding open problems in mathematics like the [cycle double cover conjecture⁠(opens in a new window)](https://cdn.openai.com/pdf/04d1d1e4-bc75-476a-97cf-49055cd98d31/cdc_proof.pdf) and beaten games like Pokémon FireRed. But on ARC-AGI-3, a benchmark of 2D puzzle games, GPT‑5.6 Sol scored just 7.8%, and GPT‑5.5 could..."},"evidence_pages":[],"related_signals":[{"id":"04acdc88-d3df-4474-925e-86fb5dbdeff1","url":"https://onlylabs.fyi/signals/04acdc88-d3df-4474-925e-86fb5dbdeff1","source_url":"https://openai.com/index/chatgpt-for-academic-researchers","title":"Accelerating scientific discovery with ChatGPT for Academic Researchers","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"openai","name":"OpenAI","category":"frontier-lab"},"occurred_at":"2026-07-29T10:00:00+00:00","first_seen_at":"2026-07-29T20:00:50.381325+00:00","date_source":"rss.item_date"},{"id":"29925019-7f48-4c85-a978-520db8d863d4","url":"https://onlylabs.fyi/signals/29925019-7f48-4c85-a978-520db8d863d4","source_url":"https://openai.com/index/gpt-5-6-frontier-intelligence-efficiency","title":"How GPT-5.6 fuses frontier intelligence with frontier efficiency","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"openai","name":"OpenAI","category":"frontier-lab"},"occurred_at":"2026-07-29T00:00:00+00:00","first_seen_at":"2026-07-29T20:00:50.381325+00:00","date_source":"rss.item_date"},{"id":"6972d8be-8299-49c8-8ee7-f85884a94488","url":"https://onlylabs.fyi/signals/6972d8be-8299-49c8-8ee7-f85884a94488","source_url":"https://openai.com/index/scientific-computing-agentic-ai","title":"Scientific computing in the age of agentic AI","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"openai","name":"OpenAI","category":"frontier-lab"},"occurred_at":"2026-07-28T17:00:00+00:00","first_seen_at":"2026-07-28T20:01:42.735421+00:00","date_source":"rss.item_date"},{"id":"0e46e60e-e6dc-4669-9f1a-fe9ff6fa4f6c","url":"https://onlylabs.fyi/signals/0e46e60e-e6dc-4669-9f1a-fe9ff6fa4f6c","source_url":"https://openai.com/index/how-ai-is-expanding-what-people-do-at-work","title":"How AI is expanding what people do at work","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"openai","name":"OpenAI","category":"frontier-lab"},"occurred_at":"2026-07-27T03:30:00+00:00","first_seen_at":"2026-07-27T12:01:45.579195+00:00","date_source":"rss.item_date"},{"id":"aa1afca2-2c20-4bda-b3eb-253ec7698b63","url":"https://onlylabs.fyi/signals/aa1afca2-2c20-4bda-b3eb-253ec7698b63","source_url":"https://openai.com/index/health-in-chatgpt","title":"Launching Health in ChatGPT ","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"openai","name":"OpenAI","category":"frontier-lab"},"occurred_at":"2026-07-23T00:00:00+00:00","first_seen_at":"2026-07-23T20:00:51.590225+00:00","date_source":"rss.item_date"},{"id":"59c3f9bd-953a-4aed-b148-bd0e76f9e3bd","url":"https://onlylabs.fyi/signals/59c3f9bd-953a-4aed-b148-bd0e76f9e3bd","source_url":"https://openai.com/index/how-news-organizations-are-using-ai","title":"How news organizations are using AI to advance their vital missions","context":null,"kind":{"key":"post_published","label":"Writing"},"org":{"slug":"openai","name":"OpenAI","category":"frontier-lab"},"occurred_at":"2026-07-22T13:00:00+00:00","first_seen_at":"2026-07-22T20:01:41.281796+00:00","date_source":"rss.item_date"}]}