[{"@context":"https:\/\/schema.org\/","@type":"BlogPosting","@id":"https:\/\/blog.terabox.com\/insights\/richard-sutton-ia-aprendizaje-por-refuerzo-llm#BlogPosting","mainEntityOfPage":"https:\/\/blog.terabox.com\/insights\/richard-sutton-ia-aprendizaje-por-refuerzo-llm","headline":"Richard Sutton: IA, Aprendizaje por Refuerzo y el Futuro","name":"Richard Sutton: IA, Aprendizaje por Refuerzo y el Futuro","description":"\ud83d\udcfa V\u00eddeo de estudio recomendado hoy: https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg Richard Sutton y el Futuro de la IA: Por qu\u00e9 la Experiencia lo es TodoM\u00e1s all\u00e1 de la imitaci\u00f3n: El problema de los LLMLa Lecci\u00f3n Amarga y el C\u00f3mputo EscalamientoEl Horizonte de la... ","datePublished":"2026-08-04","dateModified":"2026-08-04","author":{"@type":"Person","@id":"https:\/\/blog.terabox.com\/author\/flextech-admin\/#Person","name":"flextech-admin","url":"https:\/\/blog.terabox.com\/author\/flextech-admin\/","image":{"@type":"ImageObject","@id":"https:\/\/secure.gravatar.com\/avatar\/ad516503a11cd5ca435acc9bb6523536?s=150&#038;d=mm&#038;r=gforcedefault=1","url":"https:\/\/secure.gravatar.com\/avatar\/ad516503a11cd5ca435acc9bb6523536?s=150&#038;d=mm&#038;r=gforcedefault=1","height":96,"width":96}},"publisher":{"@type":"Organization","name":"terabox","logo":{"@type":"ImageObject","@id":"http:\/\/blog.terabox.com\/wp-content\/uploads\/2021\/11\/logo\u4ea7\u54c1\u540d-\u7ad6\u7248.png","url":"http:\/\/blog.terabox.com\/wp-content\/uploads\/2021\/11\/logo\u4ea7\u54c1\u540d-\u7ad6\u7248.png","width":900,"height":900}},"image":{"@type":"ImageObject","@id":"https:\/\/img.youtube.com\/vi\/21EYKqUsPfg\/maxresdefault.jpg","url":"https:\/\/img.youtube.com\/vi\/21EYKqUsPfg\/maxresdefault.jpg","height":"","width":""},"url":"https:\/\/blog.terabox.com\/insights\/richard-sutton-ia-aprendizaje-por-refuerzo-llm","video":{"@context":"http:\/\/schema.org\/","@type":"VideoObject","@id":"https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg#VideoObject","contentUrl":"https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg","name":"Richard Sutton \u2013 Father of RL thinks LLMs are a dead end","description":"Richard Sutton is the father of reinforcement learning, winner of the 2024 Turing Award, and author of The Bitter Lesson. And he thinks LLMs are a dead end. After interviewing him, my steel man of Richard\u2019s position is this: LLMs aren\u2019t capable of learning on-the-job, so no matter how much we scale, we\u2019ll need *some* new architecture to enable continual learning. And once we have it, we won\u2019t need a special training phase \u2014 the agent will just learn on-the-fly, like all humans, and indeed, like all animals. This new paradigm will render our current approach with LLMs obsolete.\n\nIn our interview, I did my best to represent the view that LLMs might function as the foundation on which experiential learning can happen\u2026 Some sparks flew. A big thanks to the Alberta Machine Intelligence Institute for inviting me up to Edmonton and for letting me use their studio and equipment. Enjoy!\n\n\ud835\udc04\ud835\udc0f\ud835\udc08\ud835\udc12\ud835\udc0e\ud835\udc03\ud835\udc04 \ud835\udc0b\ud835\udc08\ud835\udc0d\ud835\udc0a\ud835\udc12\n* Transcript: https:\/\/www.dwarkesh.com\/p\/richard-sutton\n* Apple Podcasts: https:\/\/podcasts.apple.com\/us\/podcast\/richard-sutton-father-of-rl-thinks-llms-are-a-dead-end\/id1516093381?i=1000728584744\n* Spotify: https:\/\/open.spotify.com\/episode\/3zAXRCFrHPShU4MuuIx4V5?si=c9f4bf24fb4c43e3\n\n\ud835\udc12\ud835\udc0f\ud835\udc0e\ud835\udc0d\ud835\udc12\ud835\udc0e\ud835\udc11\ud835\udc12\n* Labelbox makes it possible to train AI agents in hyperrealistic RL environments. With an experienced team of applied researchers and a massive network of subject-matter experts, Labelbox ensures your training reflects important, real-world nuance. Turn your demo projects into working systems at https:\/\/labelbox.com\/dwarkesh\n\n* Gemini Deep Research is designed for thorough exploration of hard topics. For this episode, it helped me trace reinforcement learning from early policy gradients up to current-day methods, combining clear explanations with curated examples. Try it out yourself at https:\/\/gemini.google.com\/\n\n* Hudson River Trading doesn\u2019t silo their teams. Instead, HRT researchers openly trade ideas and share strategy code in a mono-repo. This means you\u2019re able to learn at incredible speed and your contributions have impact across the entire firm. Find open roles at https:\/\/hudsonrivertrading.com\/dwarkesh\n\nTo sponsor a future episode, visit\u00a0https:\/\/dwarkesh.com\/advertise\n\n\ud835\udc13\ud835\udc08\ud835\udc0c\ud835\udc04\ud835\udc12\ud835\udc13\ud835\udc00\ud835\udc0c\ud835\udc0f\ud835\udc12\n00:00:00 \u2013 Are LLMs a dead end?\n00:13:51 \u2013 Do humans do imitation learning?\n00:23:57 \u2013 The Era of Experience\n00:34:25 \u2013 Current architectures generalize poorly out of distribution\n00:42:17 \u2013 Surprises in the AI field\n00:47:28 \u2013 Will The Bitter Lesson still apply after AGI?\n00:54:35 \u2013 Succession to AI","thumbnailUrl":["https:\/\/i.ytimg.com\/vi\/21EYKqUsPfg\/default.jpg","https:\/\/i.ytimg.com\/vi\/21EYKqUsPfg\/mqdefault.jpg","https:\/\/i.ytimg.com\/vi\/21EYKqUsPfg\/hqdefault.jpg","https:\/\/i.ytimg.com\/vi\/21EYKqUsPfg\/sddefault.jpg","https:\/\/i.ytimg.com\/vi\/21EYKqUsPfg\/maxresdefault.jpg"],"uploadDate":"2025-09-26T16:01:25+00:00","duration":"PT1H7M9S","embedUrl":"https:\/\/www.youtube.com\/embed\/21EYKqUsPfg","publisher":{"@type":"Organization","@id":"https:\/\/www.youtube.com\/channel\/UCXl4i9dYBrFOabk0xGmbkRA#Organization","url":"https:\/\/www.youtube.com\/channel\/UCXl4i9dYBrFOabk0xGmbkRA","name":"Dwarkesh Patel","description":"Deeply researched interviews\n","logo":{"url":"https:\/\/yt3.ggpht.com\/lG-z7sTfhFIW2Ne1oXMHvXMXyZSaA02_I17gUel0GAEj7OypsSHQ7PE91Vp4bTbpm3PTIAWJdko=s800-c-k-c0x00ffffff-no-rj","width":800,"height":800,"@type":"ImageObject","@id":"https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg#VideoObject_publisher_logo_ImageObject"}},"potentialAction":{"@type":"SeekToAction","@id":"https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg#VideoObject_potentialAction","target":"https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg&t={seek_to_second_number}","startOffset-input":"required name=seek_to_second_number"},"interactionStatistic":[[{"@type":"InteractionCounter","@id":"https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg#VideoObject_interactionStatistic_WatchAction","interactionType":{"@type":"WatchAction"},"userInteractionCount":769421}],{"@type":"InteractionCounter","@id":"https:\/\/www.youtube.com\/watch?v=21EYKqUsPfg#VideoObject_interactionStatistic_LikeAction","interactionType":{"@type":"LikeAction"},"userInteractionCount":16874}]},"about":["Insights","\u300eSpanish\u300f"],"wordCount":1822},{"@context":"https:\/\/schema.org\/","@type":"BreadcrumbList","itemListElement":[{"@type":"ListItem","position":1,"name":"Insights","item":"https:\/\/blog.terabox.com\/insights\/#breadcrumbitem"},{"@type":"ListItem","position":2,"name":"Richard Sutton: IA, Aprendizaje por Refuerzo y el Futuro","item":"https:\/\/blog.terabox.com\/insights\/richard-sutton-ia-aprendizaje-por-refuerzo-llm#breadcrumbitem"}]}]