[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"portal-settings:stajic:ru":3,"public-menus:all":38,"post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:ru":205,"related:post:generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing:ru:1":2242},{"statusCode":4,"data":5,"message":37},200,{"tenantId":6,"lang":7,"defaultLang":8,"siteUrl":9,"contactEmail":10,"brandName":11,"logoUrl":12,"siteName":11,"siteDescription":13,"ogImage":10,"robotsIndex":14,"socialLinks":10,"reservedSlugs":10,"seoPolicy":15},"stajic","ru","de","https:\u002F\u002Fstajic.de",null,"Stajic Platform","\u002FLogo_Planet.svg","Stajic Portal",true,{"branding":16,"relatedContent":17,"crossDomainLinks":18},{"logoUrl":12},{"enabled":14},[19,22,25,28,31,34],{"url":20,"label":21,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Ffigure.rocks","figure.rocks",{"url":23,"label":24,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Floving.rocks","loving.rocks",{"url":26,"label":27,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.com","bazify.com",{"url":29,"label":30,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.de","bazify.de",{"url":32,"label":33,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.at","bazify.at",{"url":35,"label":36,"isActive":14,"showInFooter":14,"includeInSameAs":14},"https:\u002F\u002Fbazify.ba","bazify.ba","Portal settings resolved",[39,45],{"id":40,"name":41,"location":42,"isActive":14,"isDefault":43,"items":44},1,"main-navigation","header",false,[],{"id":46,"name":47,"location":48,"isActive":14,"isDefault":14,"items":49},4,"main-menu","sidebar",[50,66,79,93,103,118,133],{"id":51,"title":52,"url":60,"target":61,"icon":62,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":64,"portfolioId":10,"children":65},"item-18",{"de":53,"en":54,"es":55,"fr":56,"it":54,"ru":57,"sr":58,"zh":59},"Startseite","Home","Inicio","Accueil","Главная","Почетна","首页","\u002Ffull-stack-web-developer-munich-performance-seo-and-maintainable-builds","_self","i-lucide-home","page",111,[],{"id":67,"title":68,"url":75,"target":61,"icon":76,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":77,"portfolioId":10,"children":78},"item-22",{"de":69,"en":69,"es":70,"fr":69,"it":71,"ru":72,"sr":73,"zh":74},"Vision","Visión","Visione","Видение","Визија","想象","\u002Fueber-uns-webdesign-muenchen-webaplikation","i-lucide-eye",113,[],{"id":80,"title":81,"url":89,"target":61,"icon":90,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":91,"portfolioId":10,"children":92},"item-19",{"de":82,"en":83,"es":84,"fr":83,"it":85,"ru":86,"sr":87,"zh":88},"Leistungen","Services","Servicios","Servizi","Услуги","Услуге","服务","\u002Fservices-dienstleistungen-muenchen","i-lucide-wrench",116,[],{"id":94,"title":95,"url":99,"target":61,"icon":100,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":101,"portfolioId":10,"children":102},"item-23",{"de":96,"en":96,"es":96,"fr":96,"it":96,"ru":97,"sr":97,"zh":98},"Blog","Блог","博客","\u002Fblog","i-lucide-book-open",112,[],{"id":104,"title":105,"url":114,"target":61,"icon":115,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":116,"portfolioId":10,"children":117},"item-32",{"de":106,"en":107,"es":108,"fr":109,"it":110,"ru":111,"sr":112,"zh":113},"Neue Technologien","New Technologies","Nuevas tecnologías","Nouvelles technologies","Nuove tecnologie","Новые технологии","Нове технологије","新技术！","\u002Fneue-webtechnologien","i-lucide-sparkles",122,[],{"id":119,"title":120,"url":129,"target":61,"icon":130,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":131,"portfolioId":10,"children":132},"item-20",{"de":121,"en":122,"es":123,"fr":124,"it":125,"ru":126,"sr":127,"zh":128},"Kontakt","Contact us!","Contacto","Contact","Contatto","Контакт","Контактирајте нас","联系我们！","\u002Fcontact","i-lucide-mail",115,[],{"id":134,"title":135,"url":144,"target":61,"icon":145,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":147},"item-21",{"de":136,"en":137,"es":138,"fr":139,"it":140,"ru":141,"sr":142,"zh":143},"Unsere Arbeit","Our Work","Nuestro trabajo","Nos réalisations","I nostri lavori","Наши работы","Наши радови","文件夹","\u002Fportfolio","i-lucide-briefcase",114,[148,161,175,181,193],{"id":149,"title":150,"url":144,"target":61,"icon":159,"isActive":14,"type":63,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":146,"portfolioId":10,"children":160},"item-24",{"de":151,"en":152,"es":153,"fr":154,"it":155,"ru":156,"sr":157,"zh":158},"Alle Projekte","All Projects","Todos los proyectos","Tous les projets","Tutti i progetti","Все проекты","Сви пројекти","所有项目","i-lucide-grid-3x3",[],{"id":162,"title":163,"url":171,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":174},"item-29",{"de":164,"en":165,"es":166,"fr":167,"it":168,"ru":169,"sr":170,"zh":143},"Local Roots, Global Reach","Local Roots - Global Reach","Empresa local ","Entreprise locale","Azienda locale","Местная компания","Локално предузеће глобално тржиште","\u002Fportfolio\u002Flocal-roots-global-reach-communication-media-systems-for-modern-business","i-lucide-folder","custom",[],{"id":176,"title":177,"url":179,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":180},"item-28",{"de":178,"en":178,"es":178,"fr":178,"it":178,"ru":178,"sr":178,"zh":178},"Solr Suggester","\u002Fportfolio\u002Fsolr-fuzzy-suggester-und-solr-infix-suggester-abfrage-ueber-ajax-und-filterung",[],{"id":182,"title":183,"url":191,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":192},"item-27",{"de":184,"en":185,"es":186,"fr":187,"it":188,"ru":189,"sr":190,"zh":185},"Firmenwebseite SEO","Company Website SEO","Sitio web corporativo SEO","Site web d’entreprise SEO","Sito web aziendale SEO","Корпоративный сайт SEO","Пословна веб-страница SEO","\u002Fportfolio\u002Fseo-sem-branding-mobile-webseite-muenchen",[],{"id":194,"title":195,"url":203,"target":61,"icon":172,"isActive":14,"type":173,"productId":10,"categoryId":10,"shopCategoryId":10,"articleId":10,"pageId":10,"portfolioId":10,"children":204},"item-31",{"de":196,"en":197,"es":198,"fr":199,"it":200,"ru":201,"sr":202,"zh":197},"Digitalisierungsportal","Digitalization Portal","Portal de digitalización","Portail de numérisation","Portale di digitalizzazione","Портал цифровизации","Портал за дигитализацију","\u002Fportfolio\u002Fdigitalisierungsportal-archiv-museum-bibliothek-ead-lido-mets-mods",[],{"statusCode":4,"data":206,"message":2241},{"id":207,"title":208,"slug":209,"content":210,"contentJson":211,"excerpt":1067,"featuredImage":1068,"featuredImageAlt":1069,"featuredImageCaption":10,"featuredImageTitle":10,"featuredImageCopyright":10,"featuredImageAuthor":10,"featuredImageSourceUrl":10,"featuredImageLicense":10,"featuredImageIsAiGenerated":43,"status":1070,"publishedAt":1071,"createdAt":1072,"updatedAt":1073,"seoLocalePaths":1074,"categories":1083,"author":1104,"translations":1109},"481","Генеративный ИИ: модели, поиск, инструменты и приложения — это не одно и то же","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u003Cp>Генеративный ИИ — это не один компонент. Продакшн-система генеративного ИИ обычно объединяет генеративную модель с кодом приложения, который предоставляет инструкции и контекст, извлекает внешние знания при необходимости, предоставляет инструменты для чтения или изменения внешних систем, управляет состоянием выполнения и разрешениями, а также превращает результат в готовый продукт. Если рассматривать модель, поиск, инструменты, контекст, среду выполнения и приложение как одно и то же, это скрывает границы, которые определяют актуальность, безопасность, надёжность, стоимость и контроль.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Прямой ответ\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Модель генерирует; поиск находит внешние доказательства; инструменты получают доступ к данным или выполняют действия; контекст — это то, что модель может видеть для текущего вывода; среда выполнения координирует исполнение; приложение владеет правилами продукта, состоянием, разрешениями, персистентностью и пользовательским опытом.\u003C\u002Fstrong> Эти слои могут быть упакованы вместе поставщиком, но их обязанности остаются разными.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Примечание о терминологии и версии\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Эта статья определяет устойчивые архитектурные обязанности, а не один стек поставщика. Примеры текущих реализаций были перепроверены \u003Cstrong>8 октября 2026 года\u003C\u002Fstrong>. API поставщиков и названия продуктов могут меняться; границы обязанностей более стабильны, чем любой отдельный SDK или эндпоинт.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cnav class=\"editorjs-toc\" data-editorjs-toc=\"true\" aria-label=\"Содержание\">\u003Cstrong class=\"editorjs-toc__title\">Содержание\u003C\u002Fstrong>\u003Col class=\"editorjs-toc__list editorjs-toc__list--depth-0\">\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-5\" class=\"editorjs-toc__link\">Что на самом деле означает «генеративный ИИ»?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-9\" class=\"editorjs-toc__link\">Простейшая полезная модель системы генеративного ИИ\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-13\" class=\"editorjs-toc__link\">Шесть границ, которые имеют значение\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-15\" class=\"editorjs-toc__link\">1. Модель: генерация — её основная обязанность\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-19\" class=\"editorjs-toc__link\">2. Поиск: нахождение внешних доказательств — это отдельная операция\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-24\" class=\"editorjs-toc__link\">3. Инструменты: доступ и действие — это не знания модели\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-29\" class=\"editorjs-toc__link\">4. Контекст: то, что модель видит прямо сейчас\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-33\" class=\"editorjs-toc__link\">5. Среда выполнения и оркестрация: координация цикла\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-37\" class=\"editorjs-toc__link\">6. Приложение: где ИИ становится продуктом\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-41\" class=\"editorjs-toc__link\">Как части работают вместе в реальном запросе\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-45\" class=\"editorjs-toc__link\">Разные ИИ-продукты используют разные комбинации\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-48\" class=\"editorjs-toc__link\">Доказательство реализации: Aaasaasa AI Client\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-55\" class=\"editorjs-toc__link\">Типичные категориальные ошибки\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-57\" class=\"editorjs-toc__link\">Режимы отказа при разрушении границ\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-60\" class=\"editorjs-toc__link\">Что стабильно, а что зависит от версии?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-64\" class=\"editorjs-toc__link\">Тест границ компонентов ИИ\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-67\" class=\"editorjs-toc__link\">Чем генеративный ИИ не является\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-71\" class=\"editorjs-toc__link\">Куда двигаться дальше в графе знаний\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-75\" class=\"editorjs-toc__link\">Ограничения\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-79\" class=\"editorjs-toc__link\">Что могло бы изменить этот ответ?\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-82\" class=\"editorjs-toc__link\">Заключение\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-86\" class=\"editorjs-toc__link\">Часто задаваемые вопросы\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-88\" class=\"editorjs-toc__link\">Глоссарий\u003C\u002Fa>\u003C\u002Fli>\u003Cli class=\"editorjs-toc__item\">\u003Ca href=\"#section-90\" class=\"editorjs-toc__link\">Первоисточники и доказательства реализации\u003C\u002Fa>\u003C\u002Fli>\u003C\u002Fol>\u003C\u002Fnav>\n\u003Ch2 id=\"section-5\">Что на самом деле означает «генеративный ИИ»?\u003C\u002Fh2>\n\u003Cp>На уровне модели генеративный ИИ относится к моделям ИИ, которые генерируют производный синтетический контент, такой как текст, изображения, аудио, видео, код или другой цифровой вывод. NIST AI 600-1 использует это ориентированное на модель значение и отдельно обсуждает риски на уровнях модели, системы, приложения и варианта использования.\u003C\u002Fp>\n\u003Cp>Это различие важно, потому что модель ИИ — это не то же самое, что полная система ИИ. Текущий глоссарий NIST определяет модель ИИ как компонент, который производит выходные данные из входных с использованием вычислительных, статистических или машинно-обучающих методов, тогда как система ИИ может включать программное обеспечение, аппаратное обеспечение, приложения, инструменты или утилиты, которые работают с использованием ИИ.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Полезная граница\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Генеративная модель ≠ приложение генеративного ИИ.\u003C\u002Fstrong>\u003Cbr>Модель — это один вычислительный компонент. Готовый продукт ИИ — это система, построенная вокруг этого компонента.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-9\">Простейшая полезная модель системы генеративного ИИ\u003C\u002Fh2>\n\u003Cp>Для первой мысленной модели представьте корпоративного ассистента, отвечающего: «Может ли этот клиент получить возврат сегодня?» Полезный ответ может потребовать нескольких разных обязанностей. Языковая модель может интерпретировать вопрос и написать объяснение, но текущее состояние заказа может поступить из инструмента базы данных, политика возврата может поступить из поиска по документам, разрешения могут обеспечиваться приложением, а окончательное действие может потребовать контролируемого вызова API.\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Один распространённый путь выполнения\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Запрос пользователя\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Приложение получает вопрос или задачу на естественном языке.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Политика и состояние приложения\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Идентичность, арендатор, разрешения, текущее состояние рабочего процесса и правила продукта определяют, что запросу разрешено делать.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Поиск или прямой доступ к данным\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Система получает внешние доказательства или текущие факты, когда знаний модели недостаточно.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Построение контекста\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Инструкции, ввод пользователя, выбранные доказательства, релевантное состояние и определения инструментов собираются для модели.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Вывод модели\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Генеративная модель интерпретирует предоставленный контекст и производит текст, структурированный вывод или запрос инструмента.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Выполнение инструмента при необходимости\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Среда выполнения или приложение проверяет и выполняет одобренные вызовы инструментов вне модели.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Наблюдение и продолжение\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Результаты инструментов могут вернуться к модели как новый контекст для следующего шага вывода.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">8\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">8. Проверка и вывод продукта\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Приложение проверяет результат, записывает требуемое состояние или данные аудита и представляет или выполняет окончательный результат.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Реальные системы не всегда следуют этой последовательности точно. Поиск может происходить до первого вызова модели, инструменты могут выбираться во время цикла агента, детерминированная логика приложения может полностью обходить модель, а проверка может происходить на нескольких этапах. Смысл в том, чтобы разделить обязанности, а не навязать один универсальный рабочий процесс.\u003C\u002Fp>\n\u003Ch2 id=\"section-13\">Шесть границ, которые имеют значение\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Шесть обязанностей внутри одного продукта ИИ\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Основная задача\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Типичные входные данные\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Не то же самое, что\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Модель\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Поиск\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Инструменты\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Контекст\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Среда выполнения \u002F оркестратор\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Приложение\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-15\">1. Модель: генерация — её основная обязанность\u003C\u002Fh2>\n\u003Cp>Генеративная модель отображает предоставленные входные данные в сгенерированные выходные данные. Для языковой модели это может включать текст на естественном языке, структурированный JSON, код, классификации, резюме, планы или аргументы вызова инструментов. Мультимодальные генеративные модели могут работать с дополнительными типами входных и выходных данных.\u003C\u002Fp>\n\u003Cp>Модель может содержать значительные изученные знания в своих параметрах, но параметризованные знания — это не живая база данных. Модель автоматически не знает документ, созданный пять минут назад, текущий уровень запасов, частную запись клиента или состояние приложения, если эта информация не предоставлена через текущий входной путь.\u003C\u002Fp>\n\u003Cp>Именно поэтому смена модели автоматически не решает проблему устаревших знаний, отсутствующих разрешений, сломанного поиска, неправильного владения состоянием или небезопасного выполнения инструментов. Эти сбои часто относятся к другим слоям.\u003C\u002Fp>\n\u003Ch2 id=\"section-19\">2. Поиск: нахождение внешних доказательств — это отдельная операция\u003C\u002Fh2>\n\u003Cp>Поиск выбирает информацию из внешнего источника до или во время генерации. Работа 2020 года «Retrieval-Augmented Generation» Льюиса и соавторов сделала это разделение явным, объединив параметрическую генеративную модель с извлечённой непараметрической памятью. Современные продакшн-системы используют множество вариантов поиска, но архитектурная идея остаётся: полезные доказательства можно извлекать во время инференса, а не полагаться только на то, что модель выучила во время обучения.\u003C\u002Fp>\n\u003Cp>Поиск может использовать лексический поиск, эмбеддинги, векторный поиск, гибридный поиск, SQL, графы знаний, фильтры по метаданным, API или другие механизмы отбора. Таким образом, векторная база данных — это лишь один из возможных компонентов поиска, а не определение RAG.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--warning my-6 rounded-xl border p-5 border-amber-300 bg-amber-50 dark:border-amber-900 dark:bg-amber-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Релевантность — это не авторитетность\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Извлечённый фрагмент может быть крайне релевантным и при этом устаревшим, неавторизованным, из неверной версии или недостаточным для подтверждения утверждения. Качество поиска и качество доказательств необходимо оценивать отдельно.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fru\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Что такое RAG? Самое простое объяснение того, как это работает\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Каноническое объяснение retrieval-augmented generation простым языком, включая разделение между LLM, знаниями, состоянием, памятью и инструментами.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Читать основы RAG →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-24\">3. Инструменты: доступ и действие — это не знания модели\u003C\u002Fh2>\n\u003Cp>Инструмент — это интерфейс, через который среда выполнения ИИ может запросить функциональность вне модели. Инструмент может запрашивать базу данных, искать в интернете, читать файл, вычислять значение, вызывать внутренний сервис, создавать тикет, отправлять сообщение, изменять запись или запускать другую контролируемую операцию.\u003C\u002Fp>\n\u003Cp>Текущая документация OpenAI по вызову функций делает эту границу явной: вызов функций позволяет моделям взаимодействовать с внешними системами и получать доступ к данным или действиям, предоставляемым приложением. Модель может предложить или выбрать вызов, но реальную операцию выполняет внешняя система.\u003C\u002Fp>\n\u003Cp>Таким образом, использование инструментов порождает два отдельных вопроса: Может ли модель запросить эту возможность? и Авторизует и выполнит ли это приложение? Продакшн-система не должна путать намерение модели с разрешением на побочный эффект.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Намерение модели — это не полномочие на выполнение\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Модель может выдать корректный запрос к инструменту, и всё равно получить отказ. Авторизация, валидация аргументов, ограничения скорости, правила транзакций, требования аудита и откат находятся вне модели.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-29\">4. Контекст: то, что модель видит прямо сейчас\u003C\u002Fh2>\n\u003Cp>Контекст — это информация, доступная модели на конкретном шаге инференса. Руководство Anthropic по контекстной инженерии описывает контекст как набор токенов, включаемых при сэмплировании из LLM. На практике этот набор может содержать системные инструкции, сообщения пользователя, историю диалога, извлечённые доказательства, определения инструментов, результаты работы инструментов, сводки памяти и выбранное состояние приложения.\u003C\u002Fp>\n\u003Cp>Таким образом, контекст — это ни полная база знаний, ни долговременная память. Компания может хранить десять миллионов документов, тогда как в один вызов модели попадает лишь несколько фрагментов. Среда выполнения может сохранять год истории диалога, предоставляя только те части, которые нужны для текущей задачи.\u003C\u002Fp>\n\u003Cp>Контекстное окно также создаёт инженерное ограничение. Добавление большего количества текста не гарантирует лучший ответ; нерелевантная, устаревшая, противоречивая или низкоавторитетная информация может размыть доказательства, которые действительно важны.\u003C\u002Fp>\n\u003Ch2 id=\"section-33\">5. Среда выполнения и оркестрация: координация цикла\u003C\u002Fh2>\n\u003Cp>Среда выполнения или слой оркестрации координирует участие модели в задаче. В зависимости от архитектуры он может управлять сессиями, запросами к модели, обнаружением инструментов, циклами вызова инструментов, повторными попытками, передачей управления, потоковыми событиями, тайм-аутами, контрольными точками, компактификацией или средами исполнения.\u003C\u002Fp>\n\u003Cp>Некоторые среды выполнения — это тонкий код приложения вокруг API модели. Другие — полноценные агентные обвязки. Управляемая среда выполнения от вендора может владеть частью цикла, тогда как приложение по-прежнему владеет доменной истиной, авторизацией, бизнес-побочными эффектами и жизненным циклом продукта.\u003C\u002Fp>\n\u003Cp>Эта граница важна, потому что где работает среда выполнения и где выполняется инференс — это отдельные решения. Локально работающий клиент или агентный процесс всё равно может обращаться к удалённой модели, тогда как удалённое приложение может обращаться к модели, размещённой на инфраструктуре под контролем организации.\u003C\u002Fp>\n\u003Ch2 id=\"section-37\">6. Приложение: где ИИ становится продуктом\u003C\u002Fh2>\n\u003Cp>Приложение — это продуктовая граница вокруг компонентов ИИ. Оно отвечает за пользовательский опыт, доменную модель, текущее состояние, идентичность, область арендатора, разрешения, персистентность, интеграции с сервисами, валидацию, наблюдаемость, логику биллинга или квот, где это применимо, и правила, определяющие, что ИИ разрешено видеть или делать.\u003C\u002Fp>\n\u003Cp>Это слой, который превращает «модель может выдавать полезный результат» в «система может предоставлять надёжную возможность». Одна и та же модель может участвовать в приватном исследовательском ассистенте, рабочем процессе поддержки, кодовом агенте или коммерческом приложении, потому что окружающее приложение меняет данные, инструменты, политики, состояние и контракт выполнения.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--info my-6 rounded-xl border p-5 border-blue-300 bg-blue-50 dark:border-blue-900 dark:bg-blue-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Модель заменяема; продуктовая граница — нет\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Замена провайдера и модели может быть архитектурной целью. Авторитетное состояние приложения, разрешения, доменные правила, аудиторский след и пользовательский контракт нельзя просто делегировать той модели, которая выбрана в данный момент.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-41\">Как части работают вместе в реальном запросе\u003C\u002Fh2>\n\u003Cp>Рассмотрим ассистента поддержки, которого просят: «Верни деньги за заказ 4711, если он всё ещё подходит, и объясни почему». Запрос объединяет знания, текущее состояние, авторизацию, рассуждение и побочный эффект.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Потребность\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Правильный слой\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Почему\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Политика возврата\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Извлечение\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Система должна найти текущую применимую политику и сохранить её происхождение.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Статус заказа 4711\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Прямой доступ к данным\u002Fинструментам\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Текущая запись заказа — это изменчивое авторитетное состояние, а не то, что можно угадать из знаний модели.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Полномочия пользователя на возврат\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Приложение \u002F авторизация\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Разрешения должны применяться независимо от того, что запрашивает модель.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Сопоставить политику с фактами заказа\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель + контекст\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель может рассуждать на основе доказательств политики и текущего состояния заказа, предоставленных ей.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Выполнить возврат\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Инструмент + правила транзакций приложения\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Контролируемая внешняя операция изменяет реальное состояние.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Объяснить результат\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель может сгенерировать объяснение для пользователя на основе проверенных результатов.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Аудит произошедшего\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Приложение \u002F среда выполнения\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Система записывает доказательства, вызовы, решения, побочные эффекты и ошибки по мере необходимости.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Если у ассистента есть только языковая модель, он может обсуждать возвраты, но не может безопасно знать, подходит ли заказ 4711 в данный момент, или выполнить транзакцию. Если у него есть только извлечение, он может найти политику, но всё ещё не имеет актуального состояния заказа. Если у него есть инструменты без авторизации приложения, он может стать способным, но небезопасным. Надёжность возникает из композиции слоёв с явным распределением ответственности.\u003C\u002Fp>\n\u003Ch2 id=\"section-45\">Разные ИИ-продукты используют разные комбинации\u003C\u002Fh2>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Наличие модели не определяет всю архитектуру\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Извлечение\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Инструменты\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Авторитетное состояние\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Типичная возможность\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Ассистент только с моделью\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Ассистент с опорой на извлечение\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Ассистент, использующий инструменты\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Агентное приложение\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Cp>Это архитектурные паттерны, а не рейтинги зрелости. Функция только с моделью может быть правильным дизайном, когда задача не требует внешних фактов или действий. Добавление извлечения, инструментов, памяти или агентного цикла оправдано только тогда, когда задача требует этих возможностей.\u003C\u002Fp>\n\u003Ch2 id=\"section-48\">Доказательство реализации: Aaasaasa AI Client\u003C\u002Fh2>\n\u003Caside class=\"editorjs-callout editorjs-callout--note my-6 rounded-xl border p-5 border-gray-300 bg-gray-50 dark:border-gray-700 dark:bg-gray-900\u002F40\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Основное доказательство реализации\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">В следующем разделе описывается реализация, которую я создал и проверил на основе кодовой базы и архитектурной документации Aaasaasa AI Client по состоянию на \u003Cstrong>26 июля 2026 г.\u003C\u002Fstrong> Это доказательство полезности этих границ, а не утверждение, что одна реализация является универсальным стандартом или коммерчески развёрнутым корпоративным продуктом.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Cp>Aaasaasa AI Client — это локальное настольное ИИ-рабочее пространство, созданное с использованием Nuxt 4, Electron и TypeScript. Его AI Hub намеренно разделяет агента\u002Fклиента, провайдера, модель, расположение среды выполнения, разрешения и веб-клиент, вместо того чтобы рассматривать их как одну настройку «ИИ».\u003C\u002Fp>\n\u003Cp>Это разделение создаёт конкретное поведение. Direct Chat может общаться с моделями без инструментов файловой системы или оболочки. Агент Codex может использовать выбранное рабочее пространство и профиль разрешений. Ollama может обеспечивать прямой локальный вывод, в то время как LM Studio и настраиваемые конечные точки, совместимые с OpenAI, представляют другие пути провайдеров. Локально запущенный процесс Codex всё ещё может использовать облачную модель, поэтому пользовательский интерфейс и архитектура не приравнивают локальную среду выполнения к локальному выводу.\u003C\u002Fp>\n\u003Cp>Реализация также содержит поддержку Qdrant\u002Fвекторов, возможности извлечения документов и аутентифицированный брокер MCP каталога. Эти компоненты иллюстрируют ещё одну границу: инфраструктура извлечения и доступ к инструментам могут находиться в одном продукте, не становясь свойствами самой модели.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Концепция A01\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Доказательство реализации Aaasaasa AI Client\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Идентификатор модели, зависящий от провайдера, выбирается отдельно от провайдера и среды выполнения.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Провайдер\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Ollama, LM Studio, сервисы, совместимые с OpenAI, и другие пути провайдеров представлены отдельно.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Среда выполнения\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Локальное или удалённое расположение агента\u002Fсреды выполнения отслеживается независимо от модели.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Инструменты \u002F доступ\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Direct Chat не имеет инструментов файловой системы или оболочки; контролируемый доступ к каталогам предоставляется отдельно через брокер.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Разрешения\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Профили разрешений рабочего пространства — это политика приложения\u002Fсессии, а не возможность модели.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Инфраструктура извлечения\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Поддержка векторов и извлечение документов существуют как возможности данных\u002Fизвлечения, а не как функции модели.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Приложение\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Продукт на Electron\u002FNuxt координирует пользовательский интерфейс, учётные данные, провайдеров, обнаружение среды выполнения, разрешения, инструменты и взаимодействие с моделью.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Урок реализации\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">Архитектуру стало легче осмысливать, как только \u003Cstrong>модель, провайдер, среда выполнения, разрешения, инструменты, данные и клиент\u003C\u002Fstrong> перестали представляться как один выбор конфигурации. Различие операционно: оно определяет, что может работать локально, что может получать доступ к файлам, что может вызывать платный облачный вывод и какой слой владеет авторизацией.\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-55\">Типичные категориальные ошибки\u003C\u002Fh2>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Категориальная ошибка\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Что на самом деле происходит\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">«ИИ знает наши документы».\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Приложение или слой поиска делает содержимое выбранных документов доступным для модели.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">«RAG — это наша векторная база данных».\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Векторная база данных может быть одним индексом или хранилищем, используемым конвейером поиска; RAG — это паттерн поиска и генерации.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">«Модель вызвала нашу CRM».\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель сформировала запрос к инструменту; среда выполнения или приложение авторизовало и выполнило внешний вызов.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">«Это локальный ИИ, потому что десктопный агент работает локально».\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Место выполнения и место инференса — разные вещи. Локальная среда выполнения всё равно может обращаться к удалённой модели.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">«У модели есть разрешение редактировать файлы».\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Приложение или среда выполнения предоставляет возможность использования инструмента в рамках политики разрешений; разрешение не является внутренним свойством модели.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">«Больше контекста — больше знаний».\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Контекст — это конечный вход, доступный для одного инференса. Больший контекст может содержать больше шума, противоречий или устаревшей информации.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">«Чат-бот — это архитектура ИИ».\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Чат-интерфейс — это лишь один интерфейс. Система также может включать идентичность, состояние, поиск, инструменты, среду выполнения, валидацию, хранение и наблюдаемость.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Ch2 id=\"section-57\">Режимы отказа при разрушении границ\u003C\u002Fh2>\n\u003Cp>Ошибки в границах — это не просто терминологическая проблема. Они порождают конкретные производственные сбои, требующие разных исправлений.\u003C\u002Fp>\n\u003Csection class=\"editorjs-comparison my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Диагностируйте сбойный слой, прежде чем заменять модель\u003C\u002Fh3>\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left dark:border-gray-700 dark:bg-gray-900\">\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Симптом\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Вероятная проблема границ\u003C\u002Fth>\u003Cth class=\"border border-gray-300 bg-gray-50 px-4 py-3 text-left font-semibold dark:border-gray-700 dark:bg-gray-900\">Первая архитектурная проверка\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Устаревший ответ\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Отсутствует факт о компании\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Небезопасный побочный эффект\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Путаный ответ при большом объёме предоставленного текста\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Неожиданное использование облака\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-3 text-left font-semibold dark:border-gray-700\">Агент зависает или повторяется\u003C\u002Fth>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-3 dark:border-gray-700\">\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-60\">Что стабильно, а что зависит от версии?\u003C\u002Fh2>\n\u003Cp>Архитектурные различия в этой статье намеренно не привязаны к конкретному поставщику. Приведённые ниже текущие примеры — это факты реализации, которые следует перепроверять по мере развития API.\u003C\u002Fp>\n\u003Cdiv class=\"overflow-x-auto\">\u003Ctable class=\"w-full border-collapse\">\u003Cthead>\u003Ctr>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Область\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Стабильная архитектурная идея\u003C\u002Fth>\u003Cth class=\"border border-gray-300 px-4 py-2 text-left font-semibold\">Проверенный текущий пример на 8 октября 2026 г.\u003C\u002Fth>\u003C\u002Ftr>\u003C\u002Fthead>\u003Ctbody>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель ИИ против системы\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель — это компонент внутри более широкой системы\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Текущий глоссарий NIST отдельно определяет модель ИИ и систему ИИ.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">RAG\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Генерация может быть обусловлена извлечённой внешней информацией\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Формулировка Lewis et al. 2020 остаётся основополагающей ссылкой; современные методы поиска в продакшене выходят далеко за рамки одной схемы плотного индекса.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Хостинговый поиск\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Поиск может быть предоставлен как управляемый инструмент\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">OpenAI File Search в настоящее время является инструментом Responses API, который ищет в базах знаний загруженных файлов с использованием семантического и ключевого поиска.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Вызов функций и инструментов\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Модель может запрашивать определённые приложением внешние возможности\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">OpenAI в настоящее время документирует вызов функций как интерфейс к внешним системам, данным и действиям.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Инженерия контекста\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Поведение модели зависит от конечной информации, предоставленной для текущего инференса\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Текущее инженерное руководство Anthropic определяет контекст как набор токенов, включаемых при сэмплировании из LLM, и сосредоточено на отборе этого набора.\u003C\u002Ftd>\u003C\u002Ftr>\u003Ctr>\u003Ctd class=\"border border-gray-300 px-4 py-2\">API поставщиков\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">SDK, имена инструментов, формы эндпоинтов и поддерживаемые функции меняются\u003C\u002Ftd>\u003Ctd class=\"border border-gray-300 px-4 py-2\">Относитесь к документации поставщиков как к зависящей от версии, даже если граница ответственности остаётся стабильной.\u003C\u002Ftd>\u003C\u002Ftr>\u003C\u002Ftbody>\u003C\u002Ftable>\u003C\u002Fdiv>\n\u003Cp>Таким образом, статья-источник истины должна сохранять оба уровня: стабильные концепции для архитектуры и датированные доказательства для текущих реализаций. Смешение этих двух уровней заставляет статью устаревать без необходимости быстро.\u003C\u002Fp>\n\u003Ch2 id=\"section-64\">Тест границ компонентов ИИ\u003C\u002Fh2>\n\u003Cp>При оценке функции ИИ задайте следующие вопросы по порядку. Ответы показывают, какие компоненты система действительно имеет и какие обязанности всё ещё остаются неявными.\u003C\u002Fp>\n\u003Csection class=\"editorjs-process my-6\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Семь вопросов для производственного дизайна\u003C\u002Fh3>\u003Cdiv class=\"grid grid-cols-1 md:grid-cols-2 xl:grid-cols-3 gap-4\">\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">1\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">1. Что генерирует вывод?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Определите конкретную модель и модальности или структурированные выходные данные, которые она предоставляет.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">2\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">2. Какие факты являются авторитетными вне модели?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Определите документы, базы данных, API, текущее состояние и другие источники истины.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">3\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">3. Как выбирается релевантная информация?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Разделите прямой поиск, поиск, извлечение, ранжирование и построение контекста.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">4\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">4. Что может вызвать реальные побочные эффекты?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Перечислите инструменты и внешние действия, затем определите, кто их проверяет и авторизует.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">5\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">5. Что попадает в модель как контекст?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Сделайте явными инструкции, доказательства, состояние, историю, память и определения инструментов.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">6\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">6. Кто владеет циклом?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Определите среду выполнения или обвязку, которая управляет вызовами, событиями, повторными попытками, циклами инструментов и сессиями.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv class=\"editorjs-process__step min-w-0  rounded-xl border border-gray-200 dark:border-gray-700 p-4\">\u003Cdiv class=\"text-xs font-semibold text-gray-500 dark:text-gray-400\">7\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 font-semibold text-gray-900 dark:text-gray-100\">7. Что остаётся ответственностью приложения?\u003C\u002Fdiv>\u003Cdiv class=\"mt-1 text-sm text-gray-600 dark:text-gray-300\">Сделайте явными идентичность, разрешения, доменное состояние, валидацию, хранение, наблюдаемость и UX.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-67\">Чем генеративный ИИ не является\u003C\u002Fh2>\n\u003Cp>Генеративный ИИ не является синонимом LLM, хотя LLM — это крупный класс генеративных моделей. Он также не является синонимом RAG, векторной базы данных, агента, протокола инструментов, чат-интерфейса или приложения.\u003C\u002Fp>\n\u003Cp>Эти концепции могут быть связаны, но каждая отвечает на свой архитектурный вопрос. LLM отвечает на вопрос, как создаётся языковой вывод. Поиск отвечает на вопрос, откуда берутся внешние доказательства. Инструменты отвечают на вопрос, как предоставляются внешние возможности. Контекст отвечает на вопрос, что модель может видеть. Среда выполнения отвечает на вопрос, как координируется исполнение. Приложение отвечает на вопрос, как возможность становится контролируемым продуктом.\u003C\u002Fp>\n\u003Caside class=\"editorjs-callout editorjs-callout--success my-6 rounded-xl border p-5 border-emerald-300 bg-emerald-50 dark:border-emerald-900 dark:bg-emerald-950\u002F20\" role=\"note\">\u003Cstrong class=\"block mb-2 text-gray-900 dark:text-gray-100\">Если вы запомните только одну модель\u003C\u002Fstrong>\u003Cdiv class=\"text-gray-700 dark:text-gray-200\">\u003Cstrong>Модель = генерировать.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Поиск = находить доказательства.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Инструменты = читать или действовать вне модели.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Контекст = то, что модель видит сейчас.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Среда выполнения = координировать исполнение.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Приложение = владеть продуктом, состоянием, правилами и разрешениями.\u003C\u002Fstrong>\u003C\u002Fdiv>\u003C\u002Faside>\n\u003Ch2 id=\"section-71\">Куда двигаться дальше в графе знаний\u003C\u002Fh2>\n\u003Cp>Когда эти границы ясны, более глубокие темы становится легче разместить. RAG относится к поиску и построению контекста. Retrieval Trigger решает, когда требуются внешние доказательства. Память агента касается того, что сохраняется во времени. Вызов инструментов и MCP относятся к доступу к возможностям. Обвязки агентов относятся к оркестрации среды выполнения. RBAC, изоляция арендаторов и доменная авторизация относятся к границе безопасности приложения и платформы.\u003C\u002Fp>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fru\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Откуда LLM берёт данные? Источники данных RAG на Python\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Практическое продолжение, показывающее, как файлы, SQL, API, полнотекстовый поиск, эмбеддинги и сборка контекста связывают внешние данные с LLM.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Посмотреть путь данных в коде →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Caside class=\"editorjs-referral my-6\">\u003Ca href=\"https:\u002F\u002Fstajic.de\u002Fru\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger\" class=\"flex flex-col sm:flex-row gap-4 rounded-xl border border-gray-200 dark:border-gray-700 p-4 transition hover:border-primary-500\">\u003Cdiv class=\"min-w-0 flex-1\">\u003Cstrong class=\"block text-lg text-gray-900 dark:text-gray-100\">Когда ИИ должен перестать доверять собственным знаниям? — Триггер извлечения\u003C\u002Fstrong>\u003Cp class=\"mt-2 text-sm text-gray-600 dark:text-gray-300\">Модель принятия решений о том, когда ИИ-система должна перестать полагаться только на знания модели и получить внешние доказательства.\u003C\u002Fp>\u003Cspan class=\"mt-3 inline-flex text-sm font-medium text-primary-600 dark:text-primary-400\">Прочитать модель решения об извлечении →\u003C\u002Fspan>\u003C\u002Fdiv>\u003C\u002Fa>\u003C\u002Faside>\n\u003Ch2 id=\"section-75\">Ограничения\u003C\u002Fh2>\n\u003Cp>Шестиуровневая модель — это карта ответственности, а не требование, чтобы каждый продукт развёртывал шесть отдельных сервисов. Небольшое приложение может реализовать построение контекста, извлечение и оркестрацию внутри одного процесса. Управляемая платформа может объединить несколько обязанностей за одним API. Физическое развёртывание может быть совмещено, при этом семантическое владение остаётся раздельным.\u003C\u002Fp>\n\u003Cp>Терминология также различается у поставщиков и в исследованиях. «Агент», «среда выполнения», «память», «инструмент», «коннектор» и «контекст» могут определяться по-разному. Определения здесь выбраны так, чтобы сделать операционное владение и диагностику сбоев явными, а не утверждать, что каждый фреймворк использует идентичный словарь.\u003C\u002Fp>\n\u003Cp>Раздел об AI-клиенте Aaasaasa документирует один шаблон реализации. Он демонстрирует, что явные границы практичны, но не доказывает, что такая же компоновка компонентов оптимальна для каждого AI-продукта.\u003C\u002Fp>\n\u003Ch2 id=\"section-79\">Что могло бы изменить этот ответ?\u003C\u002Fh2>\n\u003Cp>Карту ответственности пришлось бы пересмотреть, если бы сами архитектуры моделей начали владеть авторитетным внешним состоянием, разрешениями, долговечными транзакционными побочными эффектами и проверяемым доступом к источникам как внутренними свойствами, а не возможностями, предоставляемыми окружающей системой. Текущие производственные архитектуры не делают это безопасным общим допущением.\u003C\u002Fp>\n\u003Cp>Отдельные примеры реализации изменятся гораздо раньше. Хостинговые инструменты извлечения, API агентов, интеграции MCP, функции управления контекстом и возможности поставщиков развиваются быстро. Эти детали следует обновлять, не разрушая базовые различия между генерацией, доказательствами, доступом к возможностям, контекстом, выполнением и управлением приложением.\u003C\u002Fp>\n\u003Ch2 id=\"section-82\">Заключение\u003C\u002Fh2>\n\u003Cp>Генеративный ИИ становится проще проектировать, как только «ИИ» перестаёт рассматриваться как один чёрный ящик. Модель — это генеративный компонент, а не полный продукт. Извлечение предоставляет внешние доказательства. Инструменты открывают возможности. Контекст переносит выбранную информацию в текущий вывод. Среда выполнения координирует исполнение. Приложение владеет авторитетной границей продукта.\u003C\u002Fp>\n\u003Cp>Это разделение полезно не только для объяснения. Оно говорит инженерам, откуда берутся устаревшие факты, где находится авторизация, почему локальная среда выполнения всё ещё может использовать облачный вывод, почему RAG не равен векторной базе данных, почему вызовы инструментов требуют валидации и почему смена модели не может исправить каждый системный сбой.\u003C\u002Fp>\n\u003Cp>Таким образом, устойчивый архитектурный вопрос — не «Какую модель ИИ мы используем?» Он таков: какую ответственность несёт каждый компонент, какие доказательства пересекают каждую границу и какому уровню разрешено изменять реальное состояние?\u003C\u002Fp>\n\u003Ch2 id=\"section-86\">Часто задаваемые вопросы\u003C\u002Fh2>\n\u003Csection class=\"editorjs-faq my-6 rounded-xl border border-gray-200 p-5 dark:border-gray-700\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Границы систем генеративного ИИ\u003C\u002Fh3>\u003Cdiv id=\"faq1\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Генеративный ИИ — это то же самое, что LLM?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Нет. LLM — это один тип генеративной модели. Генеративный ИИ также включает другие модальности, а производственная система генеративного ИИ может включать извлечение, инструменты, логику среды выполнения, состояние приложения, разрешения, персистентность и пользовательские интерфейсы вокруг модели.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq2\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Является ли RAG частью модели?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Обычно нет. RAG — это шаблон приложения\u002Fсистемы, который извлекает внешнюю информацию и предоставляет выбранные доказательства модели. Некоторые платформы тесно упаковывают извлечение с API моделей, но ответственность остаётся отдельной.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq3\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Требуется ли векторная база данных для RAG?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Нет. RAG может использовать векторный поиск, лексический поиск, гибридное извлечение, SQL, API, графы знаний или другие методы. Определяющим свойством является извлечение внешней информации для генерации, а не одна технология хранения.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq4\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Инструменты — это то же самое, что контекст?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Нет. Инструмент — это внешняя возможность. Его определение может быть представлено в контексте, и его результат может позже попасть в контекст, но фактическая возможность выполняется вне модели.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq5\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Означает ли запуск AI-клиента локально, что модель локальна?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Нет. Расположение среды выполнения и расположение вывода — это разные вещи. Локальное настольное приложение или агент может вызывать удалённую модель, а удалённое приложение может вызывать внутренне размещённую модель.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq6\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Кто должен обеспечивать разрешения для инструментов ИИ?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Авторизацию должна обеспечивать граница безопасности приложения или среды выполнения. Модель может запросить операцию, но намерение модели никогда не должно рассматриваться как достаточные полномочия на выполнение.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003Cdiv id=\"faq7\" class=\"border-t border-gray-200 py-4 first:border-t-0 dark:border-gray-700\">\u003Ch4 class=\"font-semibold text-gray-900 dark:text-gray-100\">Где должно находиться текущее состояние приложения?\u003C\u002Fh4>\u003Cdiv class=\"mt-2 text-gray-600 dark:text-gray-300\">Авторитетное изменчивое состояние обычно должно оставаться в приложении или доменной системе, которой оно принадлежит. ИИ может получать соответствующее состояние через контролируемый контекст или доступ к инструментам, когда это необходимо.\u003C\u002Fdiv>\u003C\u002Fdiv>\u003C\u002Fsection>\n\u003Ch2 id=\"section-88\">Глоссарий\u003C\u002Fh2>\n\u003Csection class=\"editorjs-glossary my-6 rounded-xl border border-gray-200 dark:border-gray-700 p-5\">\u003Ch3 class=\"mb-3 text-lg font-semibold\">Ключевые термины\u003C\u002Fh3>\u003Cdl>\u003Cdiv id=\"generative-model\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Генеративная модель\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Модель ИИ, предназначенная для генерации производного синтетического контента, такого как текст, изображения, аудио, видео, код или структурированный вывод.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"retrieval\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Извлечение\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Процесс выбора релевантной информации из внешнего источника или хранилища для текущей задачи.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"rag\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">RAG\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Retrieval-Augmented Generation: шаблон, в котором извлечённая внешняя информация предоставляется генеративной модели для улучшения текущего вывода.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"tool\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Инструмент\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Возможность, предоставляемая среде выполнения ИИ для чтения данных, вычислений, поиска или выполнения внешнего действия.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"context\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Контекст\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Информация, доступная модели для конкретного шага вывода.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"runtime-orchestrator\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Среда выполнения \u002F оркестратор\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Программный слой, который координирует вызовы модели, вызовы инструментов, циклы задач, сессии, повторные попытки, события или среды исполнения.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"application\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Приложение\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Продуктовый и доменный слой, который владеет взаимодействием с пользователем, авторитетным состоянием, разрешениями, валидацией, персистентностью и бизнес-поведением.\u003C\u002Fdd>\u003C\u002Fdiv>\u003Cdiv id=\"provider\" class=\"border-t border-gray-200 dark:border-gray-700 py-3 first:border-t-0\">\u003Cdt class=\"font-semibold text-gray-900 dark:text-gray-100\">Поставщик\u003C\u002Fdt>\u003Cdd class=\"mt-1 text-gray-600 dark:text-gray-300\">Сервис или среда выполнения, предоставляющая доступ к одной или нескольким моделям; идентичность поставщика и идентичность модели — это отдельные вопросы.\u003C\u002Fdd>\u003C\u002Fdiv>\u003C\u002Fdl>\u003C\u002Fsection>\n\u003Ch2 id=\"section-90\">Первоисточники и доказательства реализации\u003C\u002Fh2>\n\u003Cp>Приведённые ниже стабильные определения опираются на стандарты и исследования; быстро меняющиеся примеры реализации используют актуальную официальную инженерную документацию. Aaasaasa AI Client является оригинальным доказательством реализации и был проверен на соответствие состоянию его кодовой базы и документации на 26 июля 2026 года.\u003C\u002Fp>\n\u003Ca href=\"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST AI 600-1 — Профиль генеративного искусственного интеллекта\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Профиль генеративного ИИ от NIST, включающий определение генеративного ИИ и явное разграничение вопросов уровня модели, системы, приложения и варианта использования.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST — Модель искусственного интеллекта\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Текущее определение глоссария NIST для модели ИИ как компонента информационной системы, который производит выходные данные из входных с использованием методов ИИ.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">NIST — Система искусственного интеллекта\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Текущее определение глоссария NIST, показывающее, что система ИИ может включать системы данных, программное обеспечение, аппаратное обеспечение, приложения, инструменты или утилиты, использующие ИИ.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Льюис и др. — Генерация с дополнением из поиска для задач NLP, требующих знаний\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Статья 2020 года, представившая формулировку RAG, которая объединяет генеративную модель с извлечённой непараметрической памятью.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Поиск по файлам\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Текущая официальная документация по размещённому поиску файлов в Responses API с использованием баз знаний из загруженных файлов, семантического поиска и поиска по ключевым словам.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">OpenAI — Вызов функций\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Текущая официальная документация, описывающая вызов инструментов и функций как интерфейс между моделями и внешними системами, данными и действиями.\u003C\u002Fp>\u003C\u002Fa>\n\u003Ca href=\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\" target=\"_blank\" rel=\"noopener noreferrer\" class=\"editorjs-link-tool block border border-gray-200 dark:border-gray-700 rounded-lg p-4 transition text-gray-900 dark:text-gray-100 hover:border-primary-500 hover:bg-primary-50 dark:hover:bg-gray-900 hover:text-gray-900 dark:hover:text-gray-100\">\u003Cstrong class=\"block font-semibold\">Anthropic — Эффективная инженерия контекста для ИИ-агентов\u003C\u002Fstrong>\u003Cp class=\"text-sm text-gray-600 dark:text-gray-400\">Инженерное руководство, определяющее контекст как набор токенов, доступных во время сэмплирования LLM, и объясняющее, почему выбор контекста является проблемой ограниченного ресурса.\u003C\u002Fp>\u003C\u002Fa>",{"time":212,"blocks":213,"version":1066},1791475711502,[214,220,228,235,243,248,253,258,265,270,275,307,312,317,360,365,370,375,380,385,390,395,402,411,416,421,426,431,437,442,447,452,457,462,467,472,477,482,487,492,498,503,508,544,549,554,585,590,595,601,606,611,616,643,649,654,683,688,693,733,738,743,776,781,786,791,818,823,828,833,839,844,849,857,865,870,875,880,885,890,895,900,905,910,915,920,925,959,964,992,997,1002,1012,1021,1030,1039,1048,1057],{"id":215,"data":216,"type":218,"tunes":219},"intro",{"text":217},"Генеративный ИИ — это не один компонент. Продакшн-система генеративного ИИ обычно объединяет генеративную модель с кодом приложения, который предоставляет инструкции и контекст, извлекает внешние знания при необходимости, предоставляет инструменты для чтения или изменения внешних систем, управляет состоянием выполнения и разрешениями, а также превращает результат в готовый продукт. Если рассматривать модель, поиск, инструменты, контекст, среду выполнения и приложение как одно и то же, это скрывает границы, которые определяют актуальность, безопасность, надёжность, стоимость и контроль.","paragraph",{},{"id":221,"data":222,"type":226,"tunes":227},"direct",{"body":223,"title":224,"variant":225},"\u003Cstrong>Модель генерирует; поиск находит внешние доказательства; инструменты получают доступ к данным или выполняют действия; контекст — это то, что модель может видеть для текущего вывода; среда выполнения координирует исполнение; приложение владеет правилами продукта, состоянием, разрешениями, персистентностью и пользовательским опытом.\u003C\u002Fstrong> Эти слои могут быть упакованы вместе поставщиком, но их обязанности остаются разными.","Прямой ответ","info","callout",{},{"id":229,"data":230,"type":226,"tunes":234},"scope-note",{"body":231,"title":232,"variant":233},"Эта статья определяет устойчивые архитектурные обязанности, а не один стек поставщика. Примеры текущих реализаций были перепроверены \u003Cstrong>8 октября 2026 года\u003C\u002Fstrong>. API поставщиков и названия продуктов могут меняться; границы обязанностей более стабильны, чем любой отдельный SDK или эндпоинт.","Примечание о терминологии и версии","note",{},{"id":236,"data":237,"type":241,"tunes":242},"toc",{"title":238,"maxLevel":239,"minLevel":240},"Содержание",3,2,"tableOfContents",{},{"id":244,"data":245,"type":42,"tunes":247},"h-meaning",{"text":246,"level":240},"Что на самом деле означает «генеративный ИИ»?",{},{"id":249,"data":250,"type":218,"tunes":252},"p-meaning-1",{"text":251},"На уровне модели генеративный ИИ относится к моделям ИИ, которые генерируют производный синтетический контент, такой как текст, изображения, аудио, видео, код или другой цифровой вывод. NIST AI 600-1 использует это ориентированное на модель значение и отдельно обсуждает риски на уровнях модели, системы, приложения и варианта использования.",{},{"id":254,"data":255,"type":218,"tunes":257},"p-meaning-2",{"text":256},"Это различие важно, потому что модель ИИ — это не то же самое, что полная система ИИ. Текущий глоссарий NIST определяет модель ИИ как компонент, который производит выходные данные из входных с использованием вычислительных, статистических или машинно-обучающих методов, тогда как система ИИ может включать программное обеспечение, аппаратное обеспечение, приложения, инструменты или утилиты, которые работают с использованием ИИ.",{},{"id":259,"data":260,"type":226,"tunes":264},"model-system-rule",{"body":261,"title":262,"variant":263},"\u003Cstrong>Генеративная модель ≠ приложение генеративного ИИ.\u003C\u002Fstrong>\u003Cbr>Модель — это один вычислительный компонент. Готовый продукт ИИ — это система, построенная вокруг этого компонента.","Полезная граница","success",{},{"id":266,"data":267,"type":42,"tunes":269},"h-simple",{"text":268,"level":240},"Простейшая полезная модель системы генеративного ИИ",{},{"id":271,"data":272,"type":218,"tunes":274},"p-simple-1",{"text":273},"Для первой мысленной модели представьте корпоративного ассистента, отвечающего: «Может ли этот клиент получить возврат сегодня?» Полезный ответ может потребовать нескольких разных обязанностей. Языковая модель может интерпретировать вопрос и написать объяснение, но текущее состояние заказа может поступить из инструмента базы данных, политика возврата может поступить из поиска по документам, разрешения могут обеспечиваться приложением, а окончательное действие может потребовать контролируемого вызова API.",{},{"id":276,"data":277,"type":305,"tunes":306},"simple-flow",{"steps":278,"title":303,"orientation":304},[279,282,285,288,291,294,297,300],{"label":280,"description":281},"1. Запрос пользователя","Приложение получает вопрос или задачу на естественном языке.",{"label":283,"description":284},"2. Политика и состояние приложения","Идентичность, арендатор, разрешения, текущее состояние рабочего процесса и правила продукта определяют, что запросу разрешено делать.",{"label":286,"description":287},"3. Поиск или прямой доступ к данным","Система получает внешние доказательства или текущие факты, когда знаний модели недостаточно.",{"label":289,"description":290},"4. Построение контекста","Инструкции, ввод пользователя, выбранные доказательства, релевантное состояние и определения инструментов собираются для модели.",{"label":292,"description":293},"5. Вывод модели","Генеративная модель интерпретирует предоставленный контекст и производит текст, структурированный вывод или запрос инструмента.",{"label":295,"description":296},"6. Выполнение инструмента при необходимости","Среда выполнения или приложение проверяет и выполняет одобренные вызовы инструментов вне модели.",{"label":298,"description":299},"7. Наблюдение и продолжение","Результаты инструментов могут вернуться к модели как новый контекст для следующего шага вывода.",{"label":301,"description":302},"8. Проверка и вывод продукта","Приложение проверяет результат, записывает требуемое состояние или данные аудита и представляет или выполняет окончательный результат.","Один распространённый путь выполнения","auto","processFlow",{},{"id":308,"data":309,"type":218,"tunes":311},"p-simple-2",{"text":310},"Реальные системы не всегда следуют этой последовательности точно. Поиск может происходить до первого вызова модели, инструменты могут выбираться во время цикла агента, детерминированная логика приложения может полностью обходить модель, а проверка может происходить на нескольких этапах. Смысл в том, чтобы разделить обязанности, а не навязать один универсальный рабочий процесс.",{},{"id":313,"data":314,"type":42,"tunes":316},"h-boundaries",{"text":315,"level":240},"Шесть границ, которые имеют значение",{},{"id":318,"data":319,"type":358,"tunes":359},"boundary-comparison",{"rows":320,"title":346,"layout":347,"columns":348},[321,326,330,334,338,342],{"id":322,"label":323,"values":324},"model","Модель",[325,325,325],"",{"id":327,"label":328,"values":329},"retrieval","Поиск",[325,325,325],{"id":331,"label":332,"values":333},"tools","Инструменты",[325,325,325],{"id":335,"label":336,"values":337},"context","Контекст",[325,325,325],{"id":339,"label":340,"values":341},"runtime","Среда выполнения \u002F оркестратор",[325,325,325],{"id":343,"label":344,"values":345},"application","Приложение",[325,325,325],"Шесть обязанностей внутри одного продукта ИИ","table",[349,352,355],{"id":350,"label":351},"job","Основная задача",{"id":353,"label":354},"input","Типичные входные данные",{"id":356,"label":357},"not","Не то же самое, что","comparison",{},{"id":361,"data":362,"type":42,"tunes":364},"h-model",{"text":363,"level":240},"1. Модель: генерация — её основная обязанность",{},{"id":366,"data":367,"type":218,"tunes":369},"p-model-1",{"text":368},"Генеративная модель отображает предоставленные входные данные в сгенерированные выходные данные. Для языковой модели это может включать текст на естественном языке, структурированный JSON, код, классификации, резюме, планы или аргументы вызова инструментов. Мультимодальные генеративные модели могут работать с дополнительными типами входных и выходных данных.",{},{"id":371,"data":372,"type":218,"tunes":374},"p-model-2",{"text":373},"Модель может содержать значительные изученные знания в своих параметрах, но параметризованные знания — это не живая база данных. Модель автоматически не знает документ, созданный пять минут назад, текущий уровень запасов, частную запись клиента или состояние приложения, если эта информация не предоставлена через текущий входной путь.",{},{"id":376,"data":377,"type":218,"tunes":379},"p-model-3",{"text":378},"Именно поэтому смена модели автоматически не решает проблему устаревших знаний, отсутствующих разрешений, сломанного поиска, неправильного владения состоянием или небезопасного выполнения инструментов. Эти сбои часто относятся к другим слоям.",{},{"id":381,"data":382,"type":42,"tunes":384},"h-retrieval",{"text":383,"level":240},"2. Поиск: нахождение внешних доказательств — это отдельная операция",{},{"id":386,"data":387,"type":218,"tunes":389},"p-retrieval-1",{"text":388},"Поиск выбирает информацию из внешнего источника до или во время генерации. Работа 2020 года «Retrieval-Augmented Generation» Льюиса и соавторов сделала это разделение явным, объединив параметрическую генеративную модель с извлечённой непараметрической памятью. Современные продакшн-системы используют множество вариантов поиска, но архитектурная идея остаётся: полезные доказательства можно извлекать во время инференса, а не полагаться только на то, что модель выучила во время обучения.",{},{"id":391,"data":392,"type":218,"tunes":394},"p-retrieval-2",{"text":393},"Поиск может использовать лексический поиск, эмбеддинги, векторный поиск, гибридный поиск, SQL, графы знаний, фильтры по метаданным, API или другие механизмы отбора. Таким образом, векторная база данных — это лишь один из возможных компонентов поиска, а не определение RAG.",{},{"id":396,"data":397,"type":226,"tunes":401},"retrieval-rule",{"body":398,"title":399,"variant":400},"Извлечённый фрагмент может быть крайне релевантным и при этом устаревшим, неавторизованным, из неверной версии или недостаточным для подтверждения утверждения. Качество поиска и качество доказательств необходимо оценивать отдельно.","Релевантность — это не авторитетность","warning",{},{"id":403,"data":404,"type":409,"tunes":410},"ref-rag",{"url":405,"title":406,"excerpt":407,"ctaLabel":408},"https:\u002F\u002Fstajic.de\u002Fru\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","Что такое RAG? Самое простое объяснение того, как это работает","Каноническое объяснение retrieval-augmented generation простым языком, включая разделение между LLM, знаниями, состоянием, памятью и инструментами.","Читать основы RAG","referralArticle",{},{"id":412,"data":413,"type":42,"tunes":415},"h-tools",{"text":414,"level":240},"3. Инструменты: доступ и действие — это не знания модели",{},{"id":417,"data":418,"type":218,"tunes":420},"p-tools-1",{"text":419},"Инструмент — это интерфейс, через который среда выполнения ИИ может запросить функциональность вне модели. Инструмент может запрашивать базу данных, искать в интернете, читать файл, вычислять значение, вызывать внутренний сервис, создавать тикет, отправлять сообщение, изменять запись или запускать другую контролируемую операцию.",{},{"id":422,"data":423,"type":218,"tunes":425},"p-tools-2",{"text":424},"Текущая документация OpenAI по вызову функций делает эту границу явной: вызов функций позволяет моделям взаимодействовать с внешними системами и получать доступ к данным или действиям, предоставляемым приложением. Модель может предложить или выбрать вызов, но реальную операцию выполняет внешняя система.",{},{"id":427,"data":428,"type":218,"tunes":430},"p-tools-3",{"text":429},"Таким образом, использование инструментов порождает два отдельных вопроса: Может ли модель запросить эту возможность? и Авторизует и выполнит ли это приложение? Продакшн-система не должна путать намерение модели с разрешением на побочный эффект.",{},{"id":432,"data":433,"type":226,"tunes":436},"tool-rule",{"body":434,"title":435,"variant":263},"Модель может выдать корректный запрос к инструменту, и всё равно получить отказ. Авторизация, валидация аргументов, ограничения скорости, правила транзакций, требования аудита и откат находятся вне модели.","Намерение модели — это не полномочие на выполнение",{},{"id":438,"data":439,"type":42,"tunes":441},"h-context",{"text":440,"level":240},"4. Контекст: то, что модель видит прямо сейчас",{},{"id":443,"data":444,"type":218,"tunes":446},"p-context-1",{"text":445},"Контекст — это информация, доступная модели на конкретном шаге инференса. Руководство Anthropic по контекстной инженерии описывает контекст как набор токенов, включаемых при сэмплировании из LLM. На практике этот набор может содержать системные инструкции, сообщения пользователя, историю диалога, извлечённые доказательства, определения инструментов, результаты работы инструментов, сводки памяти и выбранное состояние приложения.",{},{"id":448,"data":449,"type":218,"tunes":451},"p-context-2",{"text":450},"Таким образом, контекст — это ни полная база знаний, ни долговременная память. Компания может хранить десять миллионов документов, тогда как в один вызов модели попадает лишь несколько фрагментов. Среда выполнения может сохранять год истории диалога, предоставляя только те части, которые нужны для текущей задачи.",{},{"id":453,"data":454,"type":218,"tunes":456},"p-context-3",{"text":455},"Контекстное окно также создаёт инженерное ограничение. Добавление большего количества текста не гарантирует лучший ответ; нерелевантная, устаревшая, противоречивая или низкоавторитетная информация может размыть доказательства, которые действительно важны.",{},{"id":458,"data":459,"type":42,"tunes":461},"h-runtime",{"text":460,"level":240},"5. Среда выполнения и оркестрация: координация цикла",{},{"id":463,"data":464,"type":218,"tunes":466},"p-runtime-1",{"text":465},"Среда выполнения или слой оркестрации координирует участие модели в задаче. В зависимости от архитектуры он может управлять сессиями, запросами к модели, обнаружением инструментов, циклами вызова инструментов, повторными попытками, передачей управления, потоковыми событиями, тайм-аутами, контрольными точками, компактификацией или средами исполнения.",{},{"id":468,"data":469,"type":218,"tunes":471},"p-runtime-2",{"text":470},"Некоторые среды выполнения — это тонкий код приложения вокруг API модели. Другие — полноценные агентные обвязки. Управляемая среда выполнения от вендора может владеть частью цикла, тогда как приложение по-прежнему владеет доменной истиной, авторизацией, бизнес-побочными эффектами и жизненным циклом продукта.",{},{"id":473,"data":474,"type":218,"tunes":476},"p-runtime-3",{"text":475},"Эта граница важна, потому что где работает среда выполнения и где выполняется инференс — это отдельные решения. Локально работающий клиент или агентный процесс всё равно может обращаться к удалённой модели, тогда как удалённое приложение может обращаться к модели, размещённой на инфраструктуре под контролем организации.",{},{"id":478,"data":479,"type":42,"tunes":481},"h-application",{"text":480,"level":240},"6. Приложение: где ИИ становится продуктом",{},{"id":483,"data":484,"type":218,"tunes":486},"p-app-1",{"text":485},"Приложение — это продуктовая граница вокруг компонентов ИИ. Оно отвечает за пользовательский опыт, доменную модель, текущее состояние, идентичность, область арендатора, разрешения, персистентность, интеграции с сервисами, валидацию, наблюдаемость, логику биллинга или квот, где это применимо, и правила, определяющие, что ИИ разрешено видеть или делать.",{},{"id":488,"data":489,"type":218,"tunes":491},"p-app-2",{"text":490},"Это слой, который превращает «модель может выдавать полезный результат» в «система может предоставлять надёжную возможность». Одна и та же модель может участвовать в приватном исследовательском ассистенте, рабочем процессе поддержки, кодовом агенте или коммерческом приложении, потому что окружающее приложение меняет данные, инструменты, политики, состояние и контракт выполнения.",{},{"id":493,"data":494,"type":226,"tunes":497},"app-rule",{"body":495,"title":496,"variant":225},"Замена провайдера и модели может быть архитектурной целью. Авторитетное состояние приложения, разрешения, доменные правила, аудиторский след и пользовательский контракт нельзя просто делегировать той модели, которая выбрана в данный момент.","Модель заменяема; продуктовая граница — нет",{},{"id":499,"data":500,"type":42,"tunes":502},"h-work-together",{"text":501,"level":240},"Как части работают вместе в реальном запросе",{},{"id":504,"data":505,"type":218,"tunes":507},"p-together-1",{"text":506},"Рассмотрим ассистента поддержки, которого просят: «Верни деньги за заказ 4711, если он всё ещё подходит, и объясни почему». Запрос объединяет знания, текущее состояние, авторизацию, рассуждение и побочный эффект.",{},{"id":509,"data":510,"type":347,"tunes":543},"support-table",{"content":511,"stretched":43,"withHeadings":14},[512,516,520,524,528,532,536,539],[513,514,515],"Потребность","Правильный слой","Почему",[517,518,519],"Политика возврата","Извлечение","Система должна найти текущую применимую политику и сохранить её происхождение.",[521,522,523],"Статус заказа 4711","Прямой доступ к данным\u002Fинструментам","Текущая запись заказа — это изменчивое авторитетное состояние, а не то, что можно угадать из знаний модели.",[525,526,527],"Полномочия пользователя на возврат","Приложение \u002F авторизация","Разрешения должны применяться независимо от того, что запрашивает модель.",[529,530,531],"Сопоставить политику с фактами заказа","Модель + контекст","Модель может рассуждать на основе доказательств политики и текущего состояния заказа, предоставленных ей.",[533,534,535],"Выполнить возврат","Инструмент + правила транзакций приложения","Контролируемая внешняя операция изменяет реальное состояние.",[537,323,538],"Объяснить результат","Модель может сгенерировать объяснение для пользователя на основе проверенных результатов.",[540,541,542],"Аудит произошедшего","Приложение \u002F среда выполнения","Система записывает доказательства, вызовы, решения, побочные эффекты и ошибки по мере необходимости.",{},{"id":545,"data":546,"type":218,"tunes":548},"p-together-2",{"text":547},"Если у ассистента есть только языковая модель, он может обсуждать возвраты, но не может безопасно знать, подходит ли заказ 4711 в данный момент, или выполнить транзакцию. Если у него есть только извлечение, он может найти политику, но всё ещё не имеет актуального состояния заказа. Если у него есть инструменты без авторизации приложения, он может стать способным, но небезопасным. Надёжность возникает из композиции слоёв с явным распределением ответственности.",{},{"id":550,"data":551,"type":42,"tunes":553},"h-configs",{"text":552,"level":240},"Разные ИИ-продукты используют разные комбинации",{},{"id":555,"data":556,"type":358,"tunes":584},"config-comparison",{"rows":557,"title":574,"layout":347,"columns":575},[558,562,566,570],{"id":559,"label":560,"values":561},"bare","Ассистент только с моделью",[325,325,325,325],{"id":563,"label":564,"values":565},"rag","Ассистент с опорой на извлечение",[325,325,325,325],{"id":567,"label":568,"values":569},"tool","Ассистент, использующий инструменты",[325,325,325,325],{"id":571,"label":572,"values":573},"agent","Агентное приложение",[325,325,325,325],"Наличие модели не определяет всю архитектуру",[576,577,578,581],{"id":327,"label":518},{"id":331,"label":332},{"id":579,"label":580},"state","Авторитетное состояние",{"id":582,"label":583},"result","Типичная возможность",{},{"id":586,"data":587,"type":218,"tunes":589},"p-configs-1",{"text":588},"Это архитектурные паттерны, а не рейтинги зрелости. Функция только с моделью может быть правильным дизайном, когда задача не требует внешних фактов или действий. Добавление извлечения, инструментов, памяти или агентного цикла оправдано только тогда, когда задача требует этих возможностей.",{},{"id":591,"data":592,"type":42,"tunes":594},"h-implementation",{"text":593,"level":240},"Доказательство реализации: Aaasaasa AI Client",{},{"id":596,"data":597,"type":226,"tunes":600},"implementation-scope",{"body":598,"title":599,"variant":233},"В следующем разделе описывается реализация, которую я создал и проверил на основе кодовой базы и архитектурной документации Aaasaasa AI Client по состоянию на \u003Cstrong>26 июля 2026 г.\u003C\u002Fstrong> Это доказательство полезности этих границ, а не утверждение, что одна реализация является универсальным стандартом или коммерчески развёрнутым корпоративным продуктом.","Основное доказательство реализации",{},{"id":602,"data":603,"type":218,"tunes":605},"p-impl-1",{"text":604},"Aaasaasa AI Client — это локальное настольное ИИ-рабочее пространство, созданное с использованием Nuxt 4, Electron и TypeScript. Его AI Hub намеренно разделяет агента\u002Fклиента, провайдера, модель, расположение среды выполнения, разрешения и веб-клиент, вместо того чтобы рассматривать их как одну настройку «ИИ».",{},{"id":607,"data":608,"type":218,"tunes":610},"p-impl-2",{"text":609},"Это разделение создаёт конкретное поведение. Direct Chat может общаться с моделями без инструментов файловой системы или оболочки. Агент Codex может использовать выбранное рабочее пространство и профиль разрешений. Ollama может обеспечивать прямой локальный вывод, в то время как LM Studio и настраиваемые конечные точки, совместимые с OpenAI, представляют другие пути провайдеров. Локально запущенный процесс Codex всё ещё может использовать облачную модель, поэтому пользовательский интерфейс и архитектура не приравнивают локальную среду выполнения к локальному выводу.",{},{"id":612,"data":613,"type":218,"tunes":615},"p-impl-3",{"text":614},"Реализация также содержит поддержку Qdrant\u002Fвекторов, возможности извлечения документов и аутентифицированный брокер MCP каталога. Эти компоненты иллюстрируют ещё одну границу: инфраструктура извлечения и доступ к инструментам могут находиться в одном продукте, не становясь свойствами самой модели.",{},{"id":617,"data":618,"type":347,"tunes":642},"impl-map",{"content":619,"stretched":43,"withHeadings":14},[620,623,625,628,631,634,637,640],[621,622],"Концепция A01","Доказательство реализации Aaasaasa AI Client",[323,624],"Идентификатор модели, зависящий от провайдера, выбирается отдельно от провайдера и среды выполнения.",[626,627],"Провайдер","Ollama, LM Studio, сервисы, совместимые с OpenAI, и другие пути провайдеров представлены отдельно.",[629,630],"Среда выполнения","Локальное или удалённое расположение агента\u002Fсреды выполнения отслеживается независимо от модели.",[632,633],"Инструменты \u002F доступ","Direct Chat не имеет инструментов файловой системы или оболочки; контролируемый доступ к каталогам предоставляется отдельно через брокер.",[635,636],"Разрешения","Профили разрешений рабочего пространства — это политика приложения\u002Fсессии, а не возможность модели.",[638,639],"Инфраструктура извлечения","Поддержка векторов и извлечение документов существуют как возможности данных\u002Fизвлечения, а не как функции модели.",[344,641],"Продукт на Electron\u002FNuxt координирует пользовательский интерфейс, учётные данные, провайдеров, обнаружение среды выполнения, разрешения, инструменты и взаимодействие с моделью.",{},{"id":644,"data":645,"type":226,"tunes":648},"impl-lesson",{"body":646,"title":647,"variant":263},"Архитектуру стало легче осмысливать, как только \u003Cstrong>модель, провайдер, среда выполнения, разрешения, инструменты, данные и клиент\u003C\u002Fstrong> перестали представляться как один выбор конфигурации. Различие операционно: оно определяет, что может работать локально, что может получать доступ к файлам, что может вызывать платный облачный вывод и какой слой владеет авторизацией.","Урок реализации",{},{"id":650,"data":651,"type":42,"tunes":653},"h-errors",{"text":652,"level":240},"Типичные категориальные ошибки",{},{"id":655,"data":656,"type":347,"tunes":682},"errors-table",{"content":657,"stretched":43,"withHeadings":14},[658,661,664,667,670,673,676,679],[659,660],"Категориальная ошибка","Что на самом деле происходит",[662,663],"«ИИ знает наши документы».","Приложение или слой поиска делает содержимое выбранных документов доступным для модели.",[665,666],"«RAG — это наша векторная база данных».","Векторная база данных может быть одним индексом или хранилищем, используемым конвейером поиска; RAG — это паттерн поиска и генерации.",[668,669],"«Модель вызвала нашу CRM».","Модель сформировала запрос к инструменту; среда выполнения или приложение авторизовало и выполнило внешний вызов.",[671,672],"«Это локальный ИИ, потому что десктопный агент работает локально».","Место выполнения и место инференса — разные вещи. Локальная среда выполнения всё равно может обращаться к удалённой модели.",[674,675],"«У модели есть разрешение редактировать файлы».","Приложение или среда выполнения предоставляет возможность использования инструмента в рамках политики разрешений; разрешение не является внутренним свойством модели.",[677,678],"«Больше контекста — больше знаний».","Контекст — это конечный вход, доступный для одного инференса. Больший контекст может содержать больше шума, противоречий или устаревшей информации.",[680,681],"«Чат-бот — это архитектура ИИ».","Чат-интерфейс — это лишь один интерфейс. Система также может включать идентичность, состояние, поиск, инструменты, среду выполнения, валидацию, хранение и наблюдаемость.",{},{"id":684,"data":685,"type":42,"tunes":687},"h-failures",{"text":686,"level":240},"Режимы отказа при разрушении границ",{},{"id":689,"data":690,"type":218,"tunes":692},"p-failure-intro",{"text":691},"Ошибки в границах — это не просто терминологическая проблема. Они порождают конкретные производственные сбои, требующие разных исправлений.",{},{"id":694,"data":695,"type":358,"tunes":732},"failure-comparison",{"rows":696,"title":721,"layout":347,"columns":722},[697,701,705,709,713,717],{"id":698,"label":699,"values":700},"stale","Устаревший ответ",[325,325,325],{"id":702,"label":703,"values":704},"missing","Отсутствует факт о компании",[325,325,325],{"id":706,"label":707,"values":708},"unsafe","Небезопасный побочный эффект",[325,325,325],{"id":710,"label":711,"values":712},"noise","Путаный ответ при большом объёме предоставленного текста",[325,325,325],{"id":714,"label":715,"values":716},"route","Неожиданное использование облака",[325,325,325],{"id":718,"label":719,"values":720},"loop","Агент зависает или повторяется",[325,325,325],"Диагностируйте сбойный слой, прежде чем заменять модель",[723,726,729],{"id":724,"label":725},"symptom","Симптом",{"id":727,"label":728},"likely","Вероятная проблема границ",{"id":730,"label":731},"fix","Первая архитектурная проверка",{},{"id":734,"data":735,"type":42,"tunes":737},"h-version",{"text":736,"level":240},"Что стабильно, а что зависит от версии?",{},{"id":739,"data":740,"type":218,"tunes":742},"p-version-1",{"text":741},"Архитектурные различия в этой статье намеренно не привязаны к конкретному поставщику. Приведённые ниже текущие примеры — это факты реализации, которые следует перепроверять по мере развития API.",{},{"id":744,"data":745,"type":347,"tunes":775},"version-table",{"content":746,"stretched":43,"withHeadings":14},[747,751,755,759,763,767,771],[748,749,750],"Область","Стабильная архитектурная идея","Проверенный текущий пример на 8 октября 2026 г.",[752,753,754],"Модель ИИ против системы","Модель — это компонент внутри более широкой системы","Текущий глоссарий NIST отдельно определяет модель ИИ и систему ИИ.",[756,757,758],"RAG","Генерация может быть обусловлена извлечённой внешней информацией","Формулировка Lewis et al. 2020 остаётся основополагающей ссылкой; современные методы поиска в продакшене выходят далеко за рамки одной схемы плотного индекса.",[760,761,762],"Хостинговый поиск","Поиск может быть предоставлен как управляемый инструмент","OpenAI File Search в настоящее время является инструментом Responses API, который ищет в базах знаний загруженных файлов с использованием семантического и ключевого поиска.",[764,765,766],"Вызов функций и инструментов","Модель может запрашивать определённые приложением внешние возможности","OpenAI в настоящее время документирует вызов функций как интерфейс к внешним системам, данным и действиям.",[768,769,770],"Инженерия контекста","Поведение модели зависит от конечной информации, предоставленной для текущего инференса","Текущее инженерное руководство Anthropic определяет контекст как набор токенов, включаемых при сэмплировании из LLM, и сосредоточено на отборе этого набора.",[772,773,774],"API поставщиков","SDK, имена инструментов, формы эндпоинтов и поддерживаемые функции меняются","Относитесь к документации поставщиков как к зависящей от версии, даже если граница ответственности остаётся стабильной.",{},{"id":777,"data":778,"type":218,"tunes":780},"p-version-2",{"text":779},"Таким образом, статья-источник истины должна сохранять оба уровня: стабильные концепции для архитектуры и датированные доказательства для текущих реализаций. Смешение этих двух уровней заставляет статью устаревать без необходимости быстро.",{},{"id":782,"data":783,"type":42,"tunes":785},"h-test",{"text":784,"level":240},"Тест границ компонентов ИИ",{},{"id":787,"data":788,"type":218,"tunes":790},"p-test-1",{"text":789},"При оценке функции ИИ задайте следующие вопросы по порядку. Ответы показывают, какие компоненты система действительно имеет и какие обязанности всё ещё остаются неявными.",{},{"id":792,"data":793,"type":305,"tunes":817},"boundary-test",{"steps":794,"title":816,"orientation":304},[795,798,801,804,807,810,813],{"label":796,"description":797},"1. Что генерирует вывод?","Определите конкретную модель и модальности или структурированные выходные данные, которые она предоставляет.",{"label":799,"description":800},"2. Какие факты являются авторитетными вне модели?","Определите документы, базы данных, API, текущее состояние и другие источники истины.",{"label":802,"description":803},"3. Как выбирается релевантная информация?","Разделите прямой поиск, поиск, извлечение, ранжирование и построение контекста.",{"label":805,"description":806},"4. Что может вызвать реальные побочные эффекты?","Перечислите инструменты и внешние действия, затем определите, кто их проверяет и авторизует.",{"label":808,"description":809},"5. Что попадает в модель как контекст?","Сделайте явными инструкции, доказательства, состояние, историю, память и определения инструментов.",{"label":811,"description":812},"6. Кто владеет циклом?","Определите среду выполнения или обвязку, которая управляет вызовами, событиями, повторными попытками, циклами инструментов и сессиями.",{"label":814,"description":815},"7. Что остаётся ответственностью приложения?","Сделайте явными идентичность, разрешения, доменное состояние, валидацию, хранение, наблюдаемость и UX.","Семь вопросов для производственного дизайна",{},{"id":819,"data":820,"type":42,"tunes":822},"h-not",{"text":821,"level":240},"Чем генеративный ИИ не является",{},{"id":824,"data":825,"type":218,"tunes":827},"p-not-1",{"text":826},"Генеративный ИИ не является синонимом LLM, хотя LLM — это крупный класс генеративных моделей. Он также не является синонимом RAG, векторной базы данных, агента, протокола инструментов, чат-интерфейса или приложения.",{},{"id":829,"data":830,"type":218,"tunes":832},"p-not-2",{"text":831},"Эти концепции могут быть связаны, но каждая отвечает на свой архитектурный вопрос. LLM отвечает на вопрос, как создаётся языковой вывод. Поиск отвечает на вопрос, откуда берутся внешние доказательства. Инструменты отвечают на вопрос, как предоставляются внешние возможности. Контекст отвечает на вопрос, что модель может видеть. Среда выполнения отвечает на вопрос, как координируется исполнение. Приложение отвечает на вопрос, как возможность становится контролируемым продуктом.",{},{"id":834,"data":835,"type":226,"tunes":838},"remember",{"body":836,"title":837,"variant":263},"\u003Cstrong>Модель = генерировать.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Поиск = находить доказательства.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Инструменты = читать или действовать вне модели.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Контекст = то, что модель видит сейчас.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Среда выполнения = координировать исполнение.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Приложение = владеть продуктом, состоянием, правилами и разрешениями.\u003C\u002Fstrong>","Если вы запомните только одну модель",{},{"id":840,"data":841,"type":42,"tunes":843},"h-next",{"text":842,"level":240},"Куда двигаться дальше в графе знаний",{},{"id":845,"data":846,"type":218,"tunes":848},"p-next-1",{"text":847},"Когда эти границы ясны, более глубокие темы становится легче разместить. RAG относится к поиску и построению контекста. Retrieval Trigger решает, когда требуются внешние доказательства. Память агента касается того, что сохраняется во времени. Вызов инструментов и MCP относятся к доступу к возможностям. Обвязки агентов относятся к оркестрации среды выполнения. RBAC, изоляция арендаторов и доменная авторизация относятся к границе безопасности приложения и платформы.",{},{"id":850,"data":851,"type":409,"tunes":856},"ref-data",{"url":852,"title":853,"excerpt":854,"ctaLabel":855},"https:\u002F\u002Fstajic.de\u002Fru\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","Откуда LLM берёт данные? Источники данных RAG на Python","Практическое продолжение, показывающее, как файлы, SQL, API, полнотекстовый поиск, эмбеддинги и сборка контекста связывают внешние данные с LLM.","Посмотреть путь данных в коде",{},{"id":858,"data":859,"type":409,"tunes":864},"ref-trigger",{"url":860,"title":861,"excerpt":862,"ctaLabel":863},"https:\u002F\u002Fstajic.de\u002Fru\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","Когда ИИ должен перестать доверять собственным знаниям? — Триггер извлечения","Модель принятия решений о том, когда ИИ-система должна перестать полагаться только на знания модели и получить внешние доказательства.","Прочитать модель решения об извлечении",{},{"id":866,"data":867,"type":42,"tunes":869},"h-limit",{"text":868,"level":240},"Ограничения",{},{"id":871,"data":872,"type":218,"tunes":874},"p-limit-1",{"text":873},"Шестиуровневая модель — это карта ответственности, а не требование, чтобы каждый продукт развёртывал шесть отдельных сервисов. Небольшое приложение может реализовать построение контекста, извлечение и оркестрацию внутри одного процесса. Управляемая платформа может объединить несколько обязанностей за одним API. Физическое развёртывание может быть совмещено, при этом семантическое владение остаётся раздельным.",{},{"id":876,"data":877,"type":218,"tunes":879},"p-limit-2",{"text":878},"Терминология также различается у поставщиков и в исследованиях. «Агент», «среда выполнения», «память», «инструмент», «коннектор» и «контекст» могут определяться по-разному. Определения здесь выбраны так, чтобы сделать операционное владение и диагностику сбоев явными, а не утверждать, что каждый фреймворк использует идентичный словарь.",{},{"id":881,"data":882,"type":218,"tunes":884},"p-limit-3",{"text":883},"Раздел об AI-клиенте Aaasaasa документирует один шаблон реализации. Он демонстрирует, что явные границы практичны, но не доказывает, что такая же компоновка компонентов оптимальна для каждого AI-продукта.",{},{"id":886,"data":887,"type":42,"tunes":889},"h-change",{"text":888,"level":240},"Что могло бы изменить этот ответ?",{},{"id":891,"data":892,"type":218,"tunes":894},"p-change-1",{"text":893},"Карту ответственности пришлось бы пересмотреть, если бы сами архитектуры моделей начали владеть авторитетным внешним состоянием, разрешениями, долговечными транзакционными побочными эффектами и проверяемым доступом к источникам как внутренними свойствами, а не возможностями, предоставляемыми окружающей системой. Текущие производственные архитектуры не делают это безопасным общим допущением.",{},{"id":896,"data":897,"type":218,"tunes":899},"p-change-2",{"text":898},"Отдельные примеры реализации изменятся гораздо раньше. Хостинговые инструменты извлечения, API агентов, интеграции MCP, функции управления контекстом и возможности поставщиков развиваются быстро. Эти детали следует обновлять, не разрушая базовые различия между генерацией, доказательствами, доступом к возможностям, контекстом, выполнением и управлением приложением.",{},{"id":901,"data":902,"type":42,"tunes":904},"h-conclusion",{"text":903,"level":240},"Заключение",{},{"id":906,"data":907,"type":218,"tunes":909},"p-conclusion-1",{"text":908},"Генеративный ИИ становится проще проектировать, как только «ИИ» перестаёт рассматриваться как один чёрный ящик. Модель — это генеративный компонент, а не полный продукт. Извлечение предоставляет внешние доказательства. Инструменты открывают возможности. Контекст переносит выбранную информацию в текущий вывод. Среда выполнения координирует исполнение. Приложение владеет авторитетной границей продукта.",{},{"id":911,"data":912,"type":218,"tunes":914},"p-conclusion-2",{"text":913},"Это разделение полезно не только для объяснения. Оно говорит инженерам, откуда берутся устаревшие факты, где находится авторизация, почему локальная среда выполнения всё ещё может использовать облачный вывод, почему RAG не равен векторной базе данных, почему вызовы инструментов требуют валидации и почему смена модели не может исправить каждый системный сбой.",{},{"id":916,"data":917,"type":218,"tunes":919},"p-conclusion-3",{"text":918},"Таким образом, устойчивый архитектурный вопрос — не «Какую модель ИИ мы используем?» Он таков: какую ответственность несёт каждый компонент, какие доказательства пересекают каждую границу и какому уровню разрешено изменять реальное состояние?",{},{"id":921,"data":922,"type":42,"tunes":924},"h-faq",{"text":923,"level":240},"Часто задаваемые вопросы",{},{"id":926,"data":927,"type":926,"tunes":958},"faq",{"items":928,"title":957},[929,933,937,941,945,949,953],{"id":930,"answer":931,"question":932},"faq1","Нет. LLM — это один тип генеративной модели. Генеративный ИИ также включает другие модальности, а производственная система генеративного ИИ может включать извлечение, инструменты, логику среды выполнения, состояние приложения, разрешения, персистентность и пользовательские интерфейсы вокруг модели.","Генеративный ИИ — это то же самое, что LLM?",{"id":934,"answer":935,"question":936},"faq2","Обычно нет. RAG — это шаблон приложения\u002Fсистемы, который извлекает внешнюю информацию и предоставляет выбранные доказательства модели. Некоторые платформы тесно упаковывают извлечение с API моделей, но ответственность остаётся отдельной.","Является ли RAG частью модели?",{"id":938,"answer":939,"question":940},"faq3","Нет. RAG может использовать векторный поиск, лексический поиск, гибридное извлечение, SQL, API, графы знаний или другие методы. Определяющим свойством является извлечение внешней информации для генерации, а не одна технология хранения.","Требуется ли векторная база данных для RAG?",{"id":942,"answer":943,"question":944},"faq4","Нет. Инструмент — это внешняя возможность. Его определение может быть представлено в контексте, и его результат может позже попасть в контекст, но фактическая возможность выполняется вне модели.","Инструменты — это то же самое, что контекст?",{"id":946,"answer":947,"question":948},"faq5","Нет. Расположение среды выполнения и расположение вывода — это разные вещи. Локальное настольное приложение или агент может вызывать удалённую модель, а удалённое приложение может вызывать внутренне размещённую модель.","Означает ли запуск AI-клиента локально, что модель локальна?",{"id":950,"answer":951,"question":952},"faq6","Авторизацию должна обеспечивать граница безопасности приложения или среды выполнения. Модель может запросить операцию, но намерение модели никогда не должно рассматриваться как достаточные полномочия на выполнение.","Кто должен обеспечивать разрешения для инструментов ИИ?",{"id":954,"answer":955,"question":956},"faq7","Авторитетное изменчивое состояние обычно должно оставаться в приложении или доменной системе, которой оно принадлежит. ИИ может получать соответствующее состояние через контролируемый контекст или доступ к инструментам, когда это необходимо.","Где должно находиться текущее состояние приложения?","Границы систем генеративного ИИ",{},{"id":960,"data":961,"type":42,"tunes":963},"h-glossary",{"text":962,"level":240},"Глоссарий",{},{"id":965,"data":966,"type":965,"tunes":991},"glossary",{"title":967,"entries":968},"Ключевые термины",[969,973,975,977,980,982,985,987],{"term":970,"anchor":971,"definition":972},"Генеративная модель","generative-model","Модель ИИ, предназначенная для генерации производного синтетического контента, такого как текст, изображения, аудио, видео, код или структурированный вывод.",{"term":518,"anchor":327,"definition":974},"Процесс выбора релевантной информации из внешнего источника или хранилища для текущей задачи.",{"term":756,"anchor":563,"definition":976},"Retrieval-Augmented Generation: шаблон, в котором извлечённая внешняя информация предоставляется генеративной модели для улучшения текущего вывода.",{"term":978,"anchor":567,"definition":979},"Инструмент","Возможность, предоставляемая среде выполнения ИИ для чтения данных, вычислений, поиска или выполнения внешнего действия.",{"term":336,"anchor":335,"definition":981},"Информация, доступная модели для конкретного шага вывода.",{"term":340,"anchor":983,"definition":984},"runtime-orchestrator","Программный слой, который координирует вызовы модели, вызовы инструментов, циклы задач, сессии, повторные попытки, события или среды исполнения.",{"term":344,"anchor":343,"definition":986},"Продуктовый и доменный слой, который владеет взаимодействием с пользователем, авторитетным состоянием, разрешениями, валидацией, персистентностью и бизнес-поведением.",{"term":988,"anchor":989,"definition":990},"Поставщик","provider","Сервис или среда выполнения, предоставляющая доступ к одной или нескольким моделям; идентичность поставщика и идентичность модели — это отдельные вопросы.",{},{"id":993,"data":994,"type":42,"tunes":996},"h-sources",{"text":995,"level":240},"Первоисточники и доказательства реализации",{},{"id":998,"data":999,"type":218,"tunes":1001},"p-sources-note",{"text":1000},"Приведённые ниже стабильные определения опираются на стандарты и исследования; быстро меняющиеся примеры реализации используют актуальную официальную инженерную документацию. Aaasaasa AI Client является оригинальным доказательством реализации и был проверен на соответствие состоянию его кодовой базы и документации на 26 июля 2026 года.",{},{"id":1003,"data":1004,"type":1010,"tunes":1011},"src-nist-profile",{"link":1005,"meta":1006},"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf",{"image":1007,"title":1008,"description":1009},{"url":325},"NIST AI 600-1 — Профиль генеративного искусственного интеллекта","Профиль генеративного ИИ от NIST, включающий определение генеративного ИИ и явное разграничение вопросов уровня модели, системы, приложения и варианта использования.","linkTool",{},{"id":1013,"data":1014,"type":1010,"tunes":1020},"src-nist-model",{"link":1015,"meta":1016},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model",{"image":1017,"title":1018,"description":1019},{"url":325},"NIST — Модель искусственного интеллекта","Текущее определение глоссария NIST для модели ИИ как компонента информационной системы, который производит выходные данные из входных с использованием методов ИИ.",{},{"id":1022,"data":1023,"type":1010,"tunes":1029},"src-nist-system",{"link":1024,"meta":1025},"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system",{"image":1026,"title":1027,"description":1028},{"url":325},"NIST — Система искусственного интеллекта","Текущее определение глоссария NIST, показывающее, что система ИИ может включать системы данных, программное обеспечение, аппаратное обеспечение, приложения, инструменты или утилиты, использующие ИИ.",{},{"id":1031,"data":1032,"type":1010,"tunes":1038},"src-rag-paper",{"link":1033,"meta":1034},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401",{"image":1035,"title":1036,"description":1037},{"url":325},"Льюис и др. — Генерация с дополнением из поиска для задач NLP, требующих знаний","Статья 2020 года, представившая формулировку RAG, которая объединяет генеративную модель с извлечённой непараметрической памятью.",{},{"id":1040,"data":1041,"type":1010,"tunes":1047},"src-openai-file-search",{"link":1042,"meta":1043},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search",{"image":1044,"title":1045,"description":1046},{"url":325},"OpenAI — Поиск по файлам","Текущая официальная документация по размещённому поиску файлов в Responses API с использованием баз знаний из загруженных файлов, семантического поиска и поиска по ключевым словам.",{},{"id":1049,"data":1050,"type":1010,"tunes":1056},"src-openai-functions",{"link":1051,"meta":1052},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling",{"image":1053,"title":1054,"description":1055},{"url":325},"OpenAI — Вызов функций","Текущая официальная документация, описывающая вызов инструментов и функций как интерфейс между моделями и внешними системами, данными и действиями.",{},{"id":1058,"data":1059,"type":1010,"tunes":1065},"src-anthropic-context",{"link":1060,"meta":1061},"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents",{"image":1062,"title":1063,"description":1064},{"url":325},"Anthropic — Эффективная инженерия контекста для ИИ-агентов","Инженерное руководство, определяющее контекст как набор токенов, доступных во время сэмплирования LLM, и объясняющее, почему выбор контекста является проблемой ограниченного ресурса.",{},"2.31","Генеративный ИИ — это больше, чем модель. Узнайте, как модели, поиск информации, инструменты, контекст, среды выполнения и приложения сочетаются друг с другом в производственных системах ИИ.","\u002Fuploads\u002F2026\u002F10\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz.webp","generative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing-1791475411822-pp0dvz","PUBLISHED","2026-10-08T12:00:00.000Z","2026-10-08T16:00:39.350Z","2026-10-08T16:10:51.437Z",{"en":1075,"de":1076,"sr":1077,"es":1078,"fr":1079,"it":1080,"ru":1081,"zh":1082},"\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fde\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fsr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fes\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Ffr\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fit\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fru\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing","\u002Fzh\u002Fblog\u002Fgenerative-ai-explained-models-retrieval-tools-and-applications-are-not-the-same-thing",[1084,1088,1092,1096,1100],{"id":1085,"name":1086,"slug":1087},84,"Политики и границы данных","policy-and-data",{"id":1089,"name":1090,"slug":1091},57,"Границы данных","data-boundaries",{"id":1093,"name":1094,"slug":1095},80,"Доступ и идентичность","access-and-identity",{"id":1097,"name":1098,"slug":1099},68,"Риски, контроли и доказательства","risks-and-controls",{"id":1101,"name":1102,"slug":1103},54,"Модель угроз","threat-model",{"id":1105,"login":1106,"email":1107,"displayName":1108},"20","rooth8233","aleksandar@stajic.de","Aleksandar Stajić",[1110,1813],{"lang":1111,"title":1112,"content":1113,"contentJson":1114,"excerpt":1812},"en","Generative AI Explained: Models, Retrieval, Tools and Applications Are Not the Same Thing","{\"time\":1791475413504,\"blocks\":[{\"id\":\"intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.\"},\"tunes\":{}},{\"id\":\"direct\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"Direct answer\",\"body\":\"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.\"},\"tunes\":{}},{\"id\":\"scope-note\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Terminology and version note\",\"body\":\"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.\"},\"tunes\":{}},{\"id\":\"toc\",\"type\":\"tableOfContents\",\"data\":{\"title\":\"Contents\",\"minLevel\":2,\"maxLevel\":3},\"tunes\":{}},{\"id\":\"h-meaning\",\"type\":\"header\",\"data\":{\"text\":\"What does “generative AI” actually mean?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-meaning-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.\"},\"tunes\":{}},{\"id\":\"p-meaning-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.\"},\"tunes\":{}},{\"id\":\"model-system-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"A useful boundary\",\"body\":\"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.\"},\"tunes\":{}},{\"id\":\"h-simple\",\"type\":\"header\",\"data\":{\"text\":\"The simplest useful model of a generative AI system\",\"level\":2},\"tunes\":{}},{\"id\":\"p-simple-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.\"},\"tunes\":{}},{\"id\":\"simple-flow\",\"type\":\"processFlow\",\"data\":{\"title\":\"One common execution path\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. User request\",\"description\":\"The application receives a natural-language question or task.\"},{\"label\":\"2. Application policy and state\",\"description\":\"Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.\"},{\"label\":\"3. Retrieval or direct data access\",\"description\":\"The system obtains external evidence or current facts when model knowledge is insufficient.\"},{\"label\":\"4. Context construction\",\"description\":\"Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.\"},{\"label\":\"5. Model inference\",\"description\":\"The generative model interprets the supplied context and produces text, structured output, or a tool request.\"},{\"label\":\"6. Tool execution when needed\",\"description\":\"The runtime or application validates and executes approved tool calls outside the model.\"},{\"label\":\"7. Observation and continuation\",\"description\":\"Tool results can return to the model as new context for another inference step.\"},{\"label\":\"8. Validation and product output\",\"description\":\"The application validates the result, records required state or audit data, and presents or executes the final outcome.\"}]},\"tunes\":{}},{\"id\":\"p-simple-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.\"},\"tunes\":{}},{\"id\":\"h-boundaries\",\"type\":\"header\",\"data\":{\"text\":\"The six boundaries that matter\",\"level\":2},\"tunes\":{}},{\"id\":\"boundary-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Six responsibilities inside one AI product\",\"layout\":\"table\",\"columns\":[{\"id\":\"job\",\"label\":\"Primary job\"},{\"id\":\"input\",\"label\":\"Typical inputs\"},{\"id\":\"not\",\"label\":\"Not the same as\"}],\"rows\":[{\"id\":\"model\",\"label\":\"Model\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"retrieval\",\"label\":\"Retrieval\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"tools\",\"label\":\"Tools\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"context\",\"label\":\"Context\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"runtime\",\"label\":\"Runtime \u002F orchestrator\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"application\",\"label\":\"Application\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-model\",\"type\":\"header\",\"data\":{\"text\":\"1. The model: generation is its core responsibility\",\"level\":2},\"tunes\":{}},{\"id\":\"p-model-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.\"},\"tunes\":{}},{\"id\":\"p-model-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.\"},\"tunes\":{}},{\"id\":\"p-model-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.\"},\"tunes\":{}},{\"id\":\"h-retrieval\",\"type\":\"header\",\"data\":{\"text\":\"2. Retrieval: finding external evidence is a separate operation\",\"level\":2},\"tunes\":{}},{\"id\":\"p-retrieval-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.\"},\"tunes\":{}},{\"id\":\"p-retrieval-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.\"},\"tunes\":{}},{\"id\":\"retrieval-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"warning\",\"title\":\"Relevance is not authority\",\"body\":\"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.\"},\"tunes\":{}},{\"id\":\"ref-rag\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works\",\"title\":\"What Is RAG? The Simplest Explanation of How It Works\",\"excerpt\":\"The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.\",\"ctaLabel\":\"Read the RAG foundation\"},\"tunes\":{}},{\"id\":\"h-tools\",\"type\":\"header\",\"data\":{\"text\":\"3. Tools: access and action are not model knowledge\",\"level\":2},\"tunes\":{}},{\"id\":\"p-tools-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.\"},\"tunes\":{}},{\"id\":\"p-tools-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.\"},\"tunes\":{}},{\"id\":\"p-tools-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.\"},\"tunes\":{}},{\"id\":\"tool-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Model intent is not execution authority\",\"body\":\"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.\"},\"tunes\":{}},{\"id\":\"h-context\",\"type\":\"header\",\"data\":{\"text\":\"4. Context: what the model can see right now\",\"level\":2},\"tunes\":{}},{\"id\":\"p-context-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.\"},\"tunes\":{}},{\"id\":\"p-context-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.\"},\"tunes\":{}},{\"id\":\"p-context-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.\"},\"tunes\":{}},{\"id\":\"h-runtime\",\"type\":\"header\",\"data\":{\"text\":\"5. Runtime and orchestration: coordinating the loop\",\"level\":2},\"tunes\":{}},{\"id\":\"p-runtime-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.\"},\"tunes\":{}},{\"id\":\"p-runtime-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.\"},\"tunes\":{}},{\"id\":\"p-runtime-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.\"},\"tunes\":{}},{\"id\":\"h-application\",\"type\":\"header\",\"data\":{\"text\":\"6. The application: where AI becomes a product\",\"level\":2},\"tunes\":{}},{\"id\":\"p-app-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.\"},\"tunes\":{}},{\"id\":\"p-app-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.\"},\"tunes\":{}},{\"id\":\"app-rule\",\"type\":\"callout\",\"data\":{\"variant\":\"info\",\"title\":\"The model is replaceable; the product boundary is not\",\"body\":\"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.\"},\"tunes\":{}},{\"id\":\"h-work-together\",\"type\":\"header\",\"data\":{\"text\":\"How the parts work together in a real request\",\"level\":2},\"tunes\":{}},{\"id\":\"p-together-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.\"},\"tunes\":{}},{\"id\":\"support-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Need\",\"Correct layer\",\"Why\"],[\"Refund policy\",\"Retrieval\",\"The system must find the current applicable policy and preserve its provenance.\"],[\"Order 4711 status\",\"Direct data\u002Ftool access\",\"The current order record is volatile authoritative state, not something to guess from model knowledge.\"],[\"User's authority to refund\",\"Application \u002F authorization\",\"Permissions must be enforced independently of what the model asks for.\"],[\"Interpret policy against order facts\",\"Model + context\",\"The model can reason over the policy evidence and current order state supplied to it.\"],[\"Execute refund\",\"Tool + application transaction rules\",\"A controlled external operation changes real state.\"],[\"Explain outcome\",\"Model\",\"The model can generate the user-facing explanation from validated results.\"],[\"Audit what happened\",\"Application \u002F runtime\",\"The system records evidence, calls, decisions, side effects, and errors as required.\"]]},\"tunes\":{}},{\"id\":\"p-together-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.\"},\"tunes\":{}},{\"id\":\"h-configs\",\"type\":\"header\",\"data\":{\"text\":\"Different AI products use different combinations\",\"level\":2},\"tunes\":{}},{\"id\":\"config-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"The presence of a model does not define the whole architecture\",\"layout\":\"table\",\"columns\":[{\"id\":\"retrieval\",\"label\":\"Retrieval\"},{\"id\":\"tools\",\"label\":\"Tools\"},{\"id\":\"state\",\"label\":\"Authoritative state\"},{\"id\":\"result\",\"label\":\"Typical capability\"}],\"rows\":[{\"id\":\"bare\",\"label\":\"Model-only assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"rag\",\"label\":\"Retrieval-grounded assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"tool\",\"label\":\"Tool-using assistant\",\"values\":[\"\",\"\",\"\",\"\"]},{\"id\":\"agent\",\"label\":\"Agentic application\",\"values\":[\"\",\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"p-configs-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.\"},\"tunes\":{}},{\"id\":\"h-implementation\",\"type\":\"header\",\"data\":{\"text\":\"Implementation evidence: Aaasaasa AI Client\",\"level\":2},\"tunes\":{}},{\"id\":\"implementation-scope\",\"type\":\"callout\",\"data\":{\"variant\":\"note\",\"title\":\"Primary implementation evidence\",\"body\":\"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.\"},\"tunes\":{}},{\"id\":\"p-impl-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.\"},\"tunes\":{}},{\"id\":\"p-impl-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.\"},\"tunes\":{}},{\"id\":\"p-impl-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.\"},\"tunes\":{}},{\"id\":\"impl-map\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"A01 concept\",\"Aaasaasa AI Client implementation evidence\"],[\"Model\",\"A provider-specific model identifier is selected separately from provider and runtime.\"],[\"Provider\",\"Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.\"],[\"Runtime\",\"Local or remote agent\u002Fruntime location is tracked independently of the model.\"],[\"Tools \u002F access\",\"Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.\"],[\"Permissions\",\"Workspace permission profiles are application\u002Fsession policy, not model capability.\"],[\"Retrieval infrastructure\",\"Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.\"],[\"Application\",\"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.\"]]},\"tunes\":{}},{\"id\":\"impl-lesson\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"Implementation lesson\",\"body\":\"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.\"},\"tunes\":{}},{\"id\":\"h-errors\",\"type\":\"header\",\"data\":{\"text\":\"Common category errors\",\"level\":2},\"tunes\":{}},{\"id\":\"errors-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Category error\",\"What is actually happening\"],[\"“The AI knows our documents.”\",\"The application or retrieval layer makes selected document content available to the model.\"],[\"“RAG is our vector database.”\",\"The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.\"],[\"“The model called our CRM.”\",\"The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.\"],[\"“It is local AI because the desktop agent runs locally.”\",\"Runtime location and inference location are separate. A local runtime can still invoke a remote model.\"],[\"“The model has permission to edit files.”\",\"The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.\"],[\"“More context means more knowledge.”\",\"Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.\"],[\"“The chatbot is the AI architecture.”\",\"The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.\"]]},\"tunes\":{}},{\"id\":\"h-failures\",\"type\":\"header\",\"data\":{\"text\":\"Failure modes when the boundaries collapse\",\"level\":2},\"tunes\":{}},{\"id\":\"p-failure-intro\",\"type\":\"paragraph\",\"data\":{\"text\":\"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.\"},\"tunes\":{}},{\"id\":\"failure-comparison\",\"type\":\"comparison\",\"data\":{\"title\":\"Diagnose the failing layer before replacing the model\",\"layout\":\"table\",\"columns\":[{\"id\":\"symptom\",\"label\":\"Symptom\"},{\"id\":\"likely\",\"label\":\"Likely boundary problem\"},{\"id\":\"fix\",\"label\":\"First architectural check\"}],\"rows\":[{\"id\":\"stale\",\"label\":\"Stale answer\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"missing\",\"label\":\"Missing company fact\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"unsafe\",\"label\":\"Unsafe side effect\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"noise\",\"label\":\"Confused answer with lots of supplied text\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"route\",\"label\":\"Unexpected cloud use\",\"values\":[\"\",\"\",\"\"]},{\"id\":\"loop\",\"label\":\"Agent stalls or repeats\",\"values\":[\"\",\"\",\"\"]}]},\"tunes\":{}},{\"id\":\"h-version\",\"type\":\"header\",\"data\":{\"text\":\"What is stable and what is version-sensitive?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-version-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.\"},\"tunes\":{}},{\"id\":\"version-table\",\"type\":\"table\",\"data\":{\"withHeadings\":true,\"stretched\":false,\"content\":[[\"Area\",\"Stable architectural idea\",\"Verified current example on 8 Oct 2026\"],[\"AI model vs system\",\"A model is a component inside a broader system\",\"NIST's current glossary separately defines AI model and AI system.\"],[\"RAG\",\"Generation can be conditioned on retrieved external information\",\"The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.\"],[\"Hosted retrieval\",\"Retrieval can be exposed as a managed tool\",\"OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.\"],[\"Function\u002Ftool calling\",\"A model can request application-defined external capabilities\",\"OpenAI currently documents function calling as an interface to external systems, data and actions.\"],[\"Context engineering\",\"Model behavior depends on the finite information supplied for the current inference\",\"Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.\"],[\"Vendor APIs\",\"SDKs, tool names, endpoint shapes and supported features change\",\"Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.\"]]},\"tunes\":{}},{\"id\":\"p-version-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.\"},\"tunes\":{}},{\"id\":\"h-test\",\"type\":\"header\",\"data\":{\"text\":\"The AI component-boundary test\",\"level\":2},\"tunes\":{}},{\"id\":\"p-test-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.\"},\"tunes\":{}},{\"id\":\"boundary-test\",\"type\":\"processFlow\",\"data\":{\"title\":\"Seven questions for a production design\",\"orientation\":\"auto\",\"steps\":[{\"label\":\"1. What generates the output?\",\"description\":\"Identify the exact model and the modalities or structured outputs it provides.\"},{\"label\":\"2. What facts are authoritative outside the model?\",\"description\":\"Identify documents, databases, APIs, current state and other sources of truth.\"},{\"label\":\"3. How is relevant information selected?\",\"description\":\"Separate direct lookup, search, retrieval, ranking and context construction.\"},{\"label\":\"4. What can cause real side effects?\",\"description\":\"List tools and external actions, then identify who validates and authorizes them.\"},{\"label\":\"5. What reaches the model as context?\",\"description\":\"Make instructions, evidence, state, history, memory and tool definitions explicit.\"},{\"label\":\"6. Who owns the loop?\",\"description\":\"Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.\"},{\"label\":\"7. What remains the application's responsibility?\",\"description\":\"Make identity, permissions, domain state, validation, persistence, observability and UX explicit.\"}]},\"tunes\":{}},{\"id\":\"h-not\",\"type\":\"header\",\"data\":{\"text\":\"What generative AI is not\",\"level\":2},\"tunes\":{}},{\"id\":\"p-not-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.\"},\"tunes\":{}},{\"id\":\"p-not-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.\"},\"tunes\":{}},{\"id\":\"remember\",\"type\":\"callout\",\"data\":{\"variant\":\"success\",\"title\":\"If you remember only one model\",\"body\":\"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>\"},\"tunes\":{}},{\"id\":\"h-next\",\"type\":\"header\",\"data\":{\"text\":\"Where to go next in the knowledge graph\",\"level\":2},\"tunes\":{}},{\"id\":\"p-next-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.\"},\"tunes\":{}},{\"id\":\"ref-data\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python\",\"title\":\"Where Does an LLM Get Its Data? RAG Data Sources in Python\",\"excerpt\":\"A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.\",\"ctaLabel\":\"See the data path in code\"},\"tunes\":{}},{\"id\":\"ref-trigger\",\"type\":\"referralArticle\",\"data\":{\"url\":\"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger\",\"title\":\"When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger\",\"excerpt\":\"A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.\",\"ctaLabel\":\"Read the retrieval decision model\"},\"tunes\":{}},{\"id\":\"h-limit\",\"type\":\"header\",\"data\":{\"text\":\"Limitations\",\"level\":2},\"tunes\":{}},{\"id\":\"p-limit-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.\"},\"tunes\":{}},{\"id\":\"p-limit-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.\"},\"tunes\":{}},{\"id\":\"p-limit-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.\"},\"tunes\":{}},{\"id\":\"h-change\",\"type\":\"header\",\"data\":{\"text\":\"What would change this answer?\",\"level\":2},\"tunes\":{}},{\"id\":\"p-change-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.\"},\"tunes\":{}},{\"id\":\"p-change-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.\"},\"tunes\":{}},{\"id\":\"h-conclusion\",\"type\":\"header\",\"data\":{\"text\":\"Conclusion\",\"level\":2},\"tunes\":{}},{\"id\":\"p-conclusion-1\",\"type\":\"paragraph\",\"data\":{\"text\":\"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.\"},\"tunes\":{}},{\"id\":\"p-conclusion-2\",\"type\":\"paragraph\",\"data\":{\"text\":\"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.\"},\"tunes\":{}},{\"id\":\"p-conclusion-3\",\"type\":\"paragraph\",\"data\":{\"text\":\"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?\"},\"tunes\":{}},{\"id\":\"h-faq\",\"type\":\"header\",\"data\":{\"text\":\"FAQ\",\"level\":2},\"tunes\":{}},{\"id\":\"faq\",\"type\":\"faq\",\"data\":{\"title\":\"Generative AI system boundaries\",\"items\":[{\"id\":\"faq1\",\"question\":\"Is generative AI the same as an LLM?\",\"answer\":\"No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.\"},{\"id\":\"faq2\",\"question\":\"Is RAG part of the model?\",\"answer\":\"Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.\"},{\"id\":\"faq3\",\"question\":\"Is a vector database required for RAG?\",\"answer\":\"No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.\"},{\"id\":\"faq4\",\"question\":\"Are tools the same as context?\",\"answer\":\"No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.\"},{\"id\":\"faq5\",\"question\":\"Does running an AI client locally mean the model is local?\",\"answer\":\"No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.\"},{\"id\":\"faq6\",\"question\":\"Who should enforce permissions for AI tools?\",\"answer\":\"The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.\"},{\"id\":\"faq7\",\"question\":\"Where does current application state belong?\",\"answer\":\"Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.\"}]},\"tunes\":{}},{\"id\":\"h-glossary\",\"type\":\"header\",\"data\":{\"text\":\"Glossary\",\"level\":2},\"tunes\":{}},{\"id\":\"glossary\",\"type\":\"glossary\",\"data\":{\"title\":\"Core terms\",\"entries\":[{\"term\":\"Generative model\",\"definition\":\"An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.\",\"anchor\":\"generative-model\"},{\"term\":\"Retrieval\",\"definition\":\"The process of selecting relevant information from an external source or store for the current task.\",\"anchor\":\"retrieval\"},{\"term\":\"RAG\",\"definition\":\"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.\",\"anchor\":\"rag\"},{\"term\":\"Tool\",\"definition\":\"A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.\",\"anchor\":\"tool\"},{\"term\":\"Context\",\"definition\":\"The information available to the model for a particular inference step.\",\"anchor\":\"context\"},{\"term\":\"Runtime \u002F orchestrator\",\"definition\":\"The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.\",\"anchor\":\"runtime-orchestrator\"},{\"term\":\"Application\",\"definition\":\"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.\",\"anchor\":\"application\"},{\"term\":\"Provider\",\"definition\":\"The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.\",\"anchor\":\"provider\"}]},\"tunes\":{}},{\"id\":\"h-sources\",\"type\":\"header\",\"data\":{\"text\":\"Primary sources and implementation evidence\",\"level\":2},\"tunes\":{}},{\"id\":\"p-sources-note\",\"type\":\"paragraph\",\"data\":{\"text\":\"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.\"},\"tunes\":{}},{\"id\":\"src-nist-profile\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fnvlpubs.nist.gov\u002Fnistpubs\u002Fai\u002FNIST.AI.600-1.pdf\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST AI 600-1 — Generative Artificial Intelligence Profile\",\"description\":\"NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.\"}},\"tunes\":{}},{\"id\":\"src-nist-model\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_model\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence Model\",\"description\":\"Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.\"}},\"tunes\":{}},{\"id\":\"src-nist-system\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fcsrc.nist.gov\u002Fglossary\u002Fterm\u002Fartificial_intelligence_system\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"NIST — Artificial Intelligence System\",\"description\":\"Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.\"}},\"tunes\":{}},{\"id\":\"src-rag-paper\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Farxiv.org\u002Fabs\u002F2005.11401\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\",\"description\":\"The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.\"}},\"tunes\":{}},{\"id\":\"src-openai-file-search\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ftools-file-search\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — File Search\",\"description\":\"Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.\"}},\"tunes\":{}},{\"id\":\"src-openai-functions\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffunction-calling\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"OpenAI — Function Calling\",\"description\":\"Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.\"}},\"tunes\":{}},{\"id\":\"src-anthropic-context\",\"type\":\"linkTool\",\"data\":{\"link\":\"https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Feffective-context-engineering-for-ai-agents\",\"meta\":{\"image\":{\"url\":\"\"},\"title\":\"Anthropic — Effective Context Engineering for AI Agents\",\"description\":\"Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.\"}},\"tunes\":{}}],\"version\":\"2.31.6\"}",{"time":1115,"blocks":1116,"version":1811},1791475413504,[1117,1121,1126,1131,1135,1139,1143,1147,1152,1156,1160,1189,1193,1197,1227,1231,1235,1239,1243,1247,1251,1255,1260,1267,1271,1275,1279,1283,1288,1292,1296,1300,1304,1308,1312,1316,1320,1324,1328,1332,1337,1341,1345,1379,1383,1387,1411,1415,1419,1424,1428,1432,1436,1462,1467,1471,1499,1503,1507,1537,1541,1545,1576,1580,1584,1588,1614,1618,1622,1626,1631,1635,1639,1646,1653,1657,1661,1665,1669,1673,1677,1681,1685,1689,1693,1697,1701,1727,1731,1754,1758,1762,1769,1776,1783,1790,1797,1804],{"id":215,"data":1118,"type":218,"tunes":1120},{"text":1119},"Generative AI is not one component. A production generative AI system usually combines a generative model with application code that supplies instructions and context, retrieves external knowledge when needed, exposes tools for reading or changing external systems, manages runtime state and permissions, and turns the result into a usable product. Treating the model, retrieval, tools, context, runtime, and application as the same thing hides the boundaries that determine freshness, security, reliability, cost, and control.",{},{"id":221,"data":1122,"type":226,"tunes":1125},{"body":1123,"title":1124,"variant":225},"\u003Cstrong>The model generates; retrieval finds external evidence; tools access data or perform actions; context is what the model can see for the current inference; the runtime coordinates execution; the application owns product rules, state, permissions, persistence, and user experience.\u003C\u002Fstrong> These layers can be packaged together by a vendor, but their responsibilities remain different.","Direct answer",{},{"id":229,"data":1127,"type":226,"tunes":1130},{"body":1128,"title":1129,"variant":233},"This article defines durable architectural responsibilities rather than one vendor stack. Current implementation examples were re-checked on \u003Cstrong>8 October 2026\u003C\u002Fstrong>. Vendor APIs and product names can change; the responsibility boundaries are more stable than any individual SDK or endpoint.","Terminology and version note",{},{"id":236,"data":1132,"type":241,"tunes":1134},{"title":1133,"maxLevel":239,"minLevel":240},"Contents",{},{"id":244,"data":1136,"type":42,"tunes":1138},{"text":1137,"level":240},"What does “generative AI” actually mean?",{},{"id":249,"data":1140,"type":218,"tunes":1142},{"text":1141},"At the model level, generative AI refers to AI models that generate derived synthetic content such as text, images, audio, video, code, or other digital output. NIST AI 600-1 uses this model-oriented meaning and separately discusses risks at model, system, application, and use-case levels.",{},{"id":254,"data":1144,"type":218,"tunes":1146},{"text":1145},"That distinction matters because an AI model is not the same thing as the complete AI system. NIST's current glossary defines an AI model as a component that produces outputs from inputs using computational, statistical, or machine-learning techniques, while an AI system can include software, hardware, applications, tools, or utilities that operate using AI.",{},{"id":259,"data":1148,"type":226,"tunes":1151},{"body":1149,"title":1150,"variant":263},"\u003Cstrong>Generative model ≠ generative AI application.\u003C\u002Fstrong>\u003Cbr>A model is one computational component. A usable AI product is a system built around that component.","A useful boundary",{},{"id":266,"data":1153,"type":42,"tunes":1155},{"text":1154,"level":240},"The simplest useful model of a generative AI system",{},{"id":271,"data":1157,"type":218,"tunes":1159},{"text":1158},"For a first mental model, imagine a company assistant answering: “Can this customer receive a refund today?” A useful answer may require several different responsibilities. The language model can interpret the question and write the explanation, but the current order state may come from a database tool, the refund policy may come from document retrieval, permissions may be enforced by the application, and the final action may require a controlled API call.",{},{"id":276,"data":1161,"type":305,"tunes":1188},{"steps":1162,"title":1187,"orientation":304},[1163,1166,1169,1172,1175,1178,1181,1184],{"label":1164,"description":1165},"1. User request","The application receives a natural-language question or task.",{"label":1167,"description":1168},"2. Application policy and state","Identity, tenant, permissions, current workflow state, and product rules define what the request is allowed to do.",{"label":1170,"description":1171},"3. Retrieval or direct data access","The system obtains external evidence or current facts when model knowledge is insufficient.",{"label":1173,"description":1174},"4. Context construction","Instructions, user input, selected evidence, relevant state, and tool definitions are assembled for the model.",{"label":1176,"description":1177},"5. Model inference","The generative model interprets the supplied context and produces text, structured output, or a tool request.",{"label":1179,"description":1180},"6. Tool execution when needed","The runtime or application validates and executes approved tool calls outside the model.",{"label":1182,"description":1183},"7. Observation and continuation","Tool results can return to the model as new context for another inference step.",{"label":1185,"description":1186},"8. Validation and product output","The application validates the result, records required state or audit data, and presents or executes the final outcome.","One common execution path",{},{"id":308,"data":1190,"type":218,"tunes":1192},{"text":1191},"Real systems do not always follow this sequence exactly. Retrieval can happen before the first model call, tools can be selected during an agent loop, deterministic application logic can bypass the model entirely, and validation can occur at several stages. The point is to separate responsibilities, not to impose one universal workflow.",{},{"id":313,"data":1194,"type":42,"tunes":1196},{"text":1195,"level":240},"The six boundaries that matter",{},{"id":318,"data":1198,"type":358,"tunes":1226},{"rows":1199,"title":1218,"layout":347,"columns":1219},[1200,1203,1206,1209,1212,1215],{"id":322,"label":1201,"values":1202},"Model",[325,325,325],{"id":327,"label":1204,"values":1205},"Retrieval",[325,325,325],{"id":331,"label":1207,"values":1208},"Tools",[325,325,325],{"id":335,"label":1210,"values":1211},"Context",[325,325,325],{"id":339,"label":1213,"values":1214},"Runtime \u002F orchestrator",[325,325,325],{"id":343,"label":1216,"values":1217},"Application",[325,325,325],"Six responsibilities inside one AI product",[1220,1222,1224],{"id":350,"label":1221},"Primary job",{"id":353,"label":1223},"Typical inputs",{"id":356,"label":1225},"Not the same as",{},{"id":361,"data":1228,"type":42,"tunes":1230},{"text":1229,"level":240},"1. The model: generation is its core responsibility",{},{"id":366,"data":1232,"type":218,"tunes":1234},{"text":1233},"A generative model maps supplied inputs to generated outputs. For a language model, that can include natural-language text, structured JSON, code, classifications, summaries, plans, or tool-call arguments. Multimodal generative models can work with additional input and output types.",{},{"id":371,"data":1236,"type":218,"tunes":1238},{"text":1237},"The model can contain substantial learned knowledge in its parameters, but parameterized knowledge is not a live database. The model does not automatically know a document created five minutes ago, the current stock level, a private customer record, or the state of an application unless that information is supplied through the current input path.",{},{"id":376,"data":1240,"type":218,"tunes":1242},{"text":1241},"This is why changing the model does not automatically solve stale knowledge, missing permissions, broken retrieval, incorrect state ownership, or unsafe tool execution. Those failures often belong to other layers.",{},{"id":381,"data":1244,"type":42,"tunes":1246},{"text":1245,"level":240},"2. Retrieval: finding external evidence is a separate operation",{},{"id":386,"data":1248,"type":218,"tunes":1250},{"text":1249},"Retrieval selects information from an external source before or during generation. The 2020 Retrieval-Augmented Generation work by Lewis et al. made the separation explicit by combining a parametric generative model with retrieved non-parametric memory. Modern production systems use many retrieval variants, but the architectural idea remains: useful evidence can be fetched at inference time instead of relying only on what the model learned during training.",{},{"id":391,"data":1252,"type":218,"tunes":1254},{"text":1253},"Retrieval can use lexical search, embeddings, vector search, hybrid search, SQL, knowledge graphs, metadata filters, APIs, or other selection mechanisms. A vector database is therefore one possible retrieval component, not the definition of RAG.",{},{"id":396,"data":1256,"type":226,"tunes":1259},{"body":1257,"title":1258,"variant":400},"A retrieved passage can be highly relevant and still be stale, unauthorized, from the wrong version, or insufficient to support a claim. Retrieval quality and evidence quality must be evaluated separately.","Relevance is not authority",{},{"id":403,"data":1261,"type":409,"tunes":1266},{"url":1262,"title":1263,"excerpt":1264,"ctaLabel":1265},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhat-is-rag-the-simplest-explanation-of-how-it-works","What Is RAG? The Simplest Explanation of How It Works","The canonical plain-English explanation of retrieval-augmented generation, including the separation between LLM, knowledge, state, memory and tools.","Read the RAG foundation",{},{"id":412,"data":1268,"type":42,"tunes":1270},{"text":1269,"level":240},"3. Tools: access and action are not model knowledge",{},{"id":417,"data":1272,"type":218,"tunes":1274},{"text":1273},"A tool is an interface through which an AI runtime can request functionality outside the model. A tool can query a database, search the web, read a file, calculate a value, call an internal service, create a ticket, send a message, modify a record, or trigger another controlled operation.",{},{"id":422,"data":1276,"type":218,"tunes":1278},{"text":1277},"OpenAI's current function-calling documentation makes this boundary explicit: function calling lets models interface with external systems and access data or actions provided by the application. The model can propose or select a call, but the external system performs the real operation.",{},{"id":427,"data":1280,"type":218,"tunes":1282},{"text":1281},"Tool use therefore creates two separate questions: Can the model request this capability? and Will the application authorize and execute it? A production system should not confuse model intent with permission to cause a side effect.",{},{"id":432,"data":1284,"type":226,"tunes":1287},{"body":1285,"title":1286,"variant":263},"A model can emit a valid tool request and still be denied. Authorization, argument validation, rate limits, transaction rules, audit requirements, and rollback belong outside the model.","Model intent is not execution authority",{},{"id":438,"data":1289,"type":42,"tunes":1291},{"text":1290,"level":240},"4. Context: what the model can see right now",{},{"id":443,"data":1293,"type":218,"tunes":1295},{"text":1294},"Context is the information available to the model for a particular inference step. Anthropic's context-engineering guidance describes context as the set of tokens included when sampling from an LLM. In practice, that set can contain system instructions, user messages, conversation history, retrieved evidence, tool definitions, tool results, memory summaries, and selected application state.",{},{"id":448,"data":1297,"type":218,"tunes":1299},{"text":1298},"Context is therefore neither the complete knowledge base nor long-term memory. A company may store ten million documents while only a handful of passages enter one model call. A runtime may persist a year of conversation history while exposing only the pieces needed for the current task.",{},{"id":453,"data":1301,"type":218,"tunes":1303},{"text":1302},"The context window also creates an engineering constraint. Adding more text does not guarantee a better answer; irrelevant, stale, contradictory, or low-authority information can dilute the evidence that actually matters.",{},{"id":458,"data":1305,"type":42,"tunes":1307},{"text":1306,"level":240},"5. Runtime and orchestration: coordinating the loop",{},{"id":463,"data":1309,"type":218,"tunes":1311},{"text":1310},"The runtime or orchestration layer coordinates how the model participates in a task. Depending on the architecture, it can manage sessions, model requests, tool discovery, tool-call loops, retries, handoffs, streaming events, timeouts, checkpoints, compaction, or execution environments.",{},{"id":468,"data":1313,"type":218,"tunes":1315},{"text":1314},"Some runtimes are thin application code around a model API. Others are full agent harnesses. A managed vendor runtime can own part of the loop while the application still owns domain truth, authorization, business side effects, and product lifecycle.",{},{"id":473,"data":1317,"type":218,"tunes":1319},{"text":1318},"This boundary is important because where the runtime runs and where inference runs are separate decisions. A locally running client or agent process can still call a remote model, while a remote application can call a model hosted on infrastructure under the organization's control.",{},{"id":478,"data":1321,"type":42,"tunes":1323},{"text":1322,"level":240},"6. The application: where AI becomes a product",{},{"id":483,"data":1325,"type":218,"tunes":1327},{"text":1326},"The application is the product boundary around the AI components. It owns the user experience, domain model, current state, identity, tenant scope, permissions, persistence, service integrations, validation, observability, billing or quota logic where relevant, and the rules that determine what the AI is allowed to see or do.",{},{"id":488,"data":1329,"type":218,"tunes":1331},{"text":1330},"This is the layer that turns “a model can produce useful output” into “a system can deliver a reliable capability.” The same model can participate in a private research assistant, a support workflow, a code agent, or a commerce application because the surrounding application changes the data, tools, policies, state, and execution contract.",{},{"id":493,"data":1333,"type":226,"tunes":1336},{"body":1334,"title":1335,"variant":225},"Provider and model substitution can be an architectural goal. The application's authoritative state, permissions, domain rules, audit trail, and user contract cannot simply be delegated to whichever model is currently selected.","The model is replaceable; the product boundary is not",{},{"id":499,"data":1338,"type":42,"tunes":1340},{"text":1339,"level":240},"How the parts work together in a real request",{},{"id":504,"data":1342,"type":218,"tunes":1344},{"text":1343},"Consider a support assistant asked: “Refund order 4711 if it is still eligible, and explain why.” The request combines knowledge, current state, authorization, reasoning, and a side effect.",{},{"id":509,"data":1346,"type":347,"tunes":1378},{"content":1347,"stretched":43,"withHeadings":14},[1348,1352,1355,1359,1363,1367,1371,1374],[1349,1350,1351],"Need","Correct layer","Why",[1353,1204,1354],"Refund policy","The system must find the current applicable policy and preserve its provenance.",[1356,1357,1358],"Order 4711 status","Direct data\u002Ftool access","The current order record is volatile authoritative state, not something to guess from model knowledge.",[1360,1361,1362],"User's authority to refund","Application \u002F authorization","Permissions must be enforced independently of what the model asks for.",[1364,1365,1366],"Interpret policy against order facts","Model + context","The model can reason over the policy evidence and current order state supplied to it.",[1368,1369,1370],"Execute refund","Tool + application transaction rules","A controlled external operation changes real state.",[1372,1201,1373],"Explain outcome","The model can generate the user-facing explanation from validated results.",[1375,1376,1377],"Audit what happened","Application \u002F runtime","The system records evidence, calls, decisions, side effects, and errors as required.",{},{"id":545,"data":1380,"type":218,"tunes":1382},{"text":1381},"If the assistant only has the language model, it can discuss refunds but cannot safely know whether order 4711 is currently eligible or perform the transaction. If it only has retrieval, it may find the policy but still lack live order state. If it has tools without application authorization, it may become capable but unsafe. Reliability comes from composing the layers with explicit ownership.",{},{"id":550,"data":1384,"type":42,"tunes":1386},{"text":1385,"level":240},"Different AI products use different combinations",{},{"id":555,"data":1388,"type":358,"tunes":1410},{"rows":1389,"title":1402,"layout":347,"columns":1403},[1390,1393,1396,1399],{"id":559,"label":1391,"values":1392},"Model-only assistant",[325,325,325,325],{"id":563,"label":1394,"values":1395},"Retrieval-grounded assistant",[325,325,325,325],{"id":567,"label":1397,"values":1398},"Tool-using assistant",[325,325,325,325],{"id":571,"label":1400,"values":1401},"Agentic application",[325,325,325,325],"The presence of a model does not define the whole architecture",[1404,1405,1406,1408],{"id":327,"label":1204},{"id":331,"label":1207},{"id":579,"label":1407},"Authoritative state",{"id":582,"label":1409},"Typical capability",{},{"id":586,"data":1412,"type":218,"tunes":1414},{"text":1413},"These are architecture patterns, not maturity rankings. A model-only feature can be the correct design when the task needs no external facts or actions. Adding retrieval, tools, memory, or an agent loop is justified only when the task requires those capabilities.",{},{"id":591,"data":1416,"type":42,"tunes":1418},{"text":1417,"level":240},"Implementation evidence: Aaasaasa AI Client",{},{"id":596,"data":1420,"type":226,"tunes":1423},{"body":1421,"title":1422,"variant":233},"The following section describes an implementation I built and reviewed against the Aaasaasa AI Client codebase and architecture documentation as of \u003Cstrong>26 July 2026\u003C\u002Fstrong>. It is evidence for the usefulness of these boundaries, not a claim that one implementation is a universal standard or a commercially deployed enterprise product.","Primary implementation evidence",{},{"id":602,"data":1425,"type":218,"tunes":1427},{"text":1426},"Aaasaasa AI Client is a local-first desktop AI workspace built with Nuxt 4, Electron and TypeScript. Its AI Hub deliberately separates agent\u002Fclient, provider, model, runtime location, permissions, and web client instead of treating them as one “AI” setting.",{},{"id":607,"data":1429,"type":218,"tunes":1431},{"text":1430},"That separation creates concrete behavior. Direct Chat can talk to models without filesystem or shell tools. A Codex agent can use a selected workspace and permission profile. Ollama can provide direct local inference, while LM Studio and configurable OpenAI-compatible endpoints represent other provider paths. A locally running Codex process can still use a cloud model, so the UI and architecture do not equate local runtime with local inference.",{},{"id":612,"data":1433,"type":218,"tunes":1435},{"text":1434},"The implementation also contains Qdrant\u002Fvector support, document-extraction capabilities and an authenticated directory MCP broker. Those components illustrate another boundary: retrieval infrastructure and tool access can live in the same product without becoming properties of the model itself.",{},{"id":617,"data":1437,"type":347,"tunes":1461},{"content":1438,"stretched":43,"withHeadings":14},[1439,1442,1444,1447,1450,1453,1456,1459],[1440,1441],"A01 concept","Aaasaasa AI Client implementation evidence",[1201,1443],"A provider-specific model identifier is selected separately from provider and runtime.",[1445,1446],"Provider","Ollama, LM Studio, OpenAI-compatible services and other provider paths are represented separately.",[1448,1449],"Runtime","Local or remote agent\u002Fruntime location is tracked independently of the model.",[1451,1452],"Tools \u002F access","Direct Chat has no filesystem or shell tools; controlled directory access is brokered separately.",[1454,1455],"Permissions","Workspace permission profiles are application\u002Fsession policy, not model capability.",[1457,1458],"Retrieval infrastructure","Vector support and document extraction exist as data\u002Fretrieval capabilities rather than model features.",[1216,1460],"The Electron\u002FNuxt product coordinates UI, credentials, providers, runtime discovery, permissions, tools and model interaction.",{},{"id":644,"data":1463,"type":226,"tunes":1466},{"body":1464,"title":1465,"variant":263},"The architecture became easier to reason about once \u003Cstrong>model, provider, runtime, permissions, tools, data and client\u003C\u002Fstrong> stopped being represented as one configuration choice. The distinction is operational: it determines what can run locally, what can access files, what may call paid cloud inference, and which layer owns authorization.","Implementation lesson",{},{"id":650,"data":1468,"type":42,"tunes":1470},{"text":1469,"level":240},"Common category errors",{},{"id":655,"data":1472,"type":347,"tunes":1498},{"content":1473,"stretched":43,"withHeadings":14},[1474,1477,1480,1483,1486,1489,1492,1495],[1475,1476],"Category error","What is actually happening",[1478,1479],"“The AI knows our documents.”","The application or retrieval layer makes selected document content available to the model.",[1481,1482],"“RAG is our vector database.”","The vector database can be one index or store used by a retrieval pipeline; RAG is the retrieval-plus-generation pattern.",[1484,1485],"“The model called our CRM.”","The model produced a tool request; the runtime\u002Fapplication authorized and executed the external call.",[1487,1488],"“It is local AI because the desktop agent runs locally.”","Runtime location and inference location are separate. A local runtime can still invoke a remote model.",[1490,1491],"“The model has permission to edit files.”","The application\u002Fruntime grants a tool capability under a permission policy; permission is not an intrinsic model property.",[1493,1494],"“More context means more knowledge.”","Context is the finite input made available for one inference. Larger context can contain more noise, conflict or stale information.",[1496,1497],"“The chatbot is the AI architecture.”","The chat UI is one interface. The system can also include identity, state, retrieval, tools, runtime, validation, persistence and observability.",{},{"id":684,"data":1500,"type":42,"tunes":1502},{"text":1501,"level":240},"Failure modes when the boundaries collapse",{},{"id":689,"data":1504,"type":218,"tunes":1506},{"text":1505},"Boundary mistakes are not merely terminology problems. They create distinct production failures that require different fixes.",{},{"id":694,"data":1508,"type":358,"tunes":1536},{"rows":1509,"title":1528,"layout":347,"columns":1529},[1510,1513,1516,1519,1522,1525],{"id":698,"label":1511,"values":1512},"Stale answer",[325,325,325],{"id":702,"label":1514,"values":1515},"Missing company fact",[325,325,325],{"id":706,"label":1517,"values":1518},"Unsafe side effect",[325,325,325],{"id":710,"label":1520,"values":1521},"Confused answer with lots of supplied text",[325,325,325],{"id":714,"label":1523,"values":1524},"Unexpected cloud use",[325,325,325],{"id":718,"label":1526,"values":1527},"Agent stalls or repeats",[325,325,325],"Diagnose the failing layer before replacing the model",[1530,1532,1534],{"id":724,"label":1531},"Symptom",{"id":727,"label":1533},"Likely boundary problem",{"id":730,"label":1535},"First architectural check",{},{"id":734,"data":1538,"type":42,"tunes":1540},{"text":1539,"level":240},"What is stable and what is version-sensitive?",{},{"id":739,"data":1542,"type":218,"tunes":1544},{"text":1543},"The architectural distinctions in this article are intentionally vendor-neutral. The current examples below are implementation facts that should be re-checked when APIs evolve.",{},{"id":744,"data":1546,"type":347,"tunes":1575},{"content":1547,"stretched":43,"withHeadings":14},[1548,1552,1556,1559,1563,1567,1571],[1549,1550,1551],"Area","Stable architectural idea","Verified current example on 8 Oct 2026",[1553,1554,1555],"AI model vs system","A model is a component inside a broader system","NIST's current glossary separately defines AI model and AI system.",[756,1557,1558],"Generation can be conditioned on retrieved external information","The Lewis et al. 2020 formulation remains the foundational reference; production retrieval methods now extend far beyond one dense index design.",[1560,1561,1562],"Hosted retrieval","Retrieval can be exposed as a managed tool","OpenAI File Search is currently a Responses API tool that searches uploaded-file knowledge bases using semantic and keyword retrieval.",[1564,1565,1566],"Function\u002Ftool calling","A model can request application-defined external capabilities","OpenAI currently documents function calling as an interface to external systems, data and actions.",[1568,1569,1570],"Context engineering","Model behavior depends on the finite information supplied for the current inference","Anthropic's current engineering guidance defines context as the token set included when sampling from the LLM and focuses on curating that set.",[1572,1573,1574],"Vendor APIs","SDKs, tool names, endpoint shapes and supported features change","Treat vendor documentation as version-sensitive even when the responsibility boundary remains stable.",{},{"id":777,"data":1577,"type":218,"tunes":1579},{"text":1578},"A source-of-truth article should therefore preserve both levels: stable concepts for architecture, and dated evidence for current implementations. Mixing the two makes an article age unnecessarily fast.",{},{"id":782,"data":1581,"type":42,"tunes":1583},{"text":1582,"level":240},"The AI component-boundary test",{},{"id":787,"data":1585,"type":218,"tunes":1587},{"text":1586},"When evaluating an AI feature, ask the following questions in order. The answers reveal which components the system actually has and which responsibilities are still implicit.",{},{"id":792,"data":1589,"type":305,"tunes":1613},{"steps":1590,"title":1612,"orientation":304},[1591,1594,1597,1600,1603,1606,1609],{"label":1592,"description":1593},"1. What generates the output?","Identify the exact model and the modalities or structured outputs it provides.",{"label":1595,"description":1596},"2. What facts are authoritative outside the model?","Identify documents, databases, APIs, current state and other sources of truth.",{"label":1598,"description":1599},"3. How is relevant information selected?","Separate direct lookup, search, retrieval, ranking and context construction.",{"label":1601,"description":1602},"4. What can cause real side effects?","List tools and external actions, then identify who validates and authorizes them.",{"label":1604,"description":1605},"5. What reaches the model as context?","Make instructions, evidence, state, history, memory and tool definitions explicit.",{"label":1607,"description":1608},"6. Who owns the loop?","Identify the runtime or harness that manages calls, events, retries, tool loops and sessions.",{"label":1610,"description":1611},"7. What remains the application's responsibility?","Make identity, permissions, domain state, validation, persistence, observability and UX explicit.","Seven questions for a production design",{},{"id":819,"data":1615,"type":42,"tunes":1617},{"text":1616,"level":240},"What generative AI is not",{},{"id":824,"data":1619,"type":218,"tunes":1621},{"text":1620},"Generative AI is not synonymous with an LLM, even though LLMs are a major class of generative model. It is also not synonymous with RAG, a vector database, an agent, a tool protocol, a chatbot UI, or an application.",{},{"id":829,"data":1623,"type":218,"tunes":1625},{"text":1624},"Those concepts can be connected, but each answers a different architectural question. An LLM asks how language output is produced. Retrieval asks where external evidence comes from. Tools ask how external capabilities are exposed. Context asks what the model can see. Runtime asks how execution is coordinated. The application asks how the capability becomes a controlled product.",{},{"id":834,"data":1627,"type":226,"tunes":1630},{"body":1628,"title":1629,"variant":263},"\u003Cstrong>Model = generate.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Retrieval = find evidence.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Tools = read or act outside the model.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Context = what the model sees now.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Runtime = coordinate execution.\u003C\u002Fstrong>\u003Cbr>\u003Cstrong>Application = own the product, state, rules and permissions.\u003C\u002Fstrong>","If you remember only one model",{},{"id":840,"data":1632,"type":42,"tunes":1634},{"text":1633,"level":240},"Where to go next in the knowledge graph",{},{"id":845,"data":1636,"type":218,"tunes":1638},{"text":1637},"Once these boundaries are clear, deeper topics become easier to place. RAG belongs in retrieval and context construction. Retrieval Trigger decides when external evidence is required. Agent memory concerns what persists across time. Tool calling and MCP belong to capability access. Agent harnesses belong to runtime orchestration. RBAC, tenant isolation and domain authorization belong to the application and platform security boundary.",{},{"id":850,"data":1640,"type":409,"tunes":1645},{"url":1641,"title":1642,"excerpt":1643,"ctaLabel":1644},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhere-does-an-llm-get-its-data-rag-data-sources-in-python","Where Does an LLM Get Its Data? RAG Data Sources in Python","A practical continuation showing how files, SQL, APIs, full-text search, embeddings and context assembly connect external data to an LLM.","See the data path in code",{},{"id":858,"data":1647,"type":409,"tunes":1652},{"url":1648,"title":1649,"excerpt":1650,"ctaLabel":1651},"https:\u002F\u002Fstajic.de\u002Fblog\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","When Should an AI Stop Trusting Its Own Knowledge? — The Retrieval Trigger","A decision model for when an AI system should stop relying only on model knowledge and obtain external evidence.","Read the retrieval decision model",{},{"id":866,"data":1654,"type":42,"tunes":1656},{"text":1655,"level":240},"Limitations",{},{"id":871,"data":1658,"type":218,"tunes":1660},{"text":1659},"The six-layer model is a responsibility map, not a requirement that every product deploy six separate services. A small application may implement context construction, retrieval and orchestration inside one process. A managed platform may bundle several responsibilities behind one API. Physical deployment can be combined while semantic ownership remains distinct.",{},{"id":876,"data":1662,"type":218,"tunes":1664},{"text":1663},"Terminology also varies across vendors and research. “Agent,” “runtime,” “memory,” “tool,” “connector,” and “context” can be defined differently. The definitions here are chosen to make operational ownership and failure diagnosis explicit rather than to claim that every framework uses identical vocabulary.",{},{"id":881,"data":1666,"type":218,"tunes":1668},{"text":1667},"The Aaasaasa AI Client section documents one implementation pattern. It demonstrates that explicit boundaries are practical, but it does not prove that the same component layout is optimal for every AI product.",{},{"id":886,"data":1670,"type":42,"tunes":1672},{"text":1671,"level":240},"What would change this answer?",{},{"id":891,"data":1674,"type":218,"tunes":1676},{"text":1675},"The responsibility map would need revision if model architectures themselves began to own authoritative external state, permissions, durable transactional side effects, and verifiable source access as intrinsic properties rather than capabilities supplied by a surrounding system. Current production architectures do not make that a safe general assumption.",{},{"id":896,"data":1678,"type":218,"tunes":1680},{"text":1679},"Individual implementation examples will change much sooner. Hosted retrieval tools, agent APIs, MCP integrations, context-management features and provider capabilities evolve quickly. Those details should be updated without collapsing the underlying distinctions between generation, evidence, capability access, context, execution and application control.",{},{"id":901,"data":1682,"type":42,"tunes":1684},{"text":1683,"level":240},"Conclusion",{},{"id":906,"data":1686,"type":218,"tunes":1688},{"text":1687},"Generative AI becomes easier to design once “the AI” stops being treated as one black box. The model is the generative component, not the complete product. Retrieval provides external evidence. Tools expose capabilities. Context carries selected information into the current inference. The runtime coordinates execution. The application owns the authoritative product boundary.",{},{"id":911,"data":1690,"type":218,"tunes":1692},{"text":1691},"That separation is useful for more than explanation. It tells engineers where stale facts originate, where authorization belongs, why a local runtime can still use cloud inference, why RAG does not equal a vector database, why tool calls require validation, and why changing the model cannot repair every system failure.",{},{"id":916,"data":1694,"type":218,"tunes":1696},{"text":1695},"The durable architecture question is therefore not “Which AI model are we using?” It is: Which responsibility does each component own, what evidence crosses each boundary, and which layer is allowed to change real state?",{},{"id":921,"data":1698,"type":42,"tunes":1700},{"text":1699,"level":240},"FAQ",{},{"id":926,"data":1702,"type":926,"tunes":1726},{"items":1703,"title":1725},[1704,1707,1710,1713,1716,1719,1722],{"id":930,"answer":1705,"question":1706},"No. An LLM is one type of generative model. Generative AI also includes other modalities, and a production generative AI system can include retrieval, tools, runtime logic, application state, permissions, persistence and user interfaces around the model.","Is generative AI the same as an LLM?",{"id":934,"answer":1708,"question":1709},"Usually no. RAG is an application\u002Fsystem pattern that retrieves external information and supplies selected evidence to the model. Some platforms package retrieval tightly with model APIs, but the responsibility remains distinct.","Is RAG part of the model?",{"id":938,"answer":1711,"question":1712},"No. RAG can use vector search, lexical search, hybrid retrieval, SQL, APIs, knowledge graphs or other methods. The defining property is retrieval of external information for generation, not one storage technology.","Is a vector database required for RAG?",{"id":942,"answer":1714,"question":1715},"No. A tool is an external capability. Its definition may be represented in context, and its result may later enter context, but the actual capability executes outside the model.","Are tools the same as context?",{"id":946,"answer":1717,"question":1718},"No. Runtime location and inference location are separate. A local desktop application or agent can call a remote model, while a remote application can call an internally hosted model.","Does running an AI client locally mean the model is local?",{"id":950,"answer":1720,"question":1721},"The application or runtime security boundary should enforce authorization. A model can request an operation, but model intent should never be treated as sufficient execution authority.","Who should enforce permissions for AI tools?",{"id":954,"answer":1723,"question":1724},"Authoritative volatile state should normally remain in the application or domain system that owns it. The AI can receive the relevant state through controlled context or tool access when needed.","Where does current application state belong?","Generative AI system boundaries",{},{"id":960,"data":1728,"type":42,"tunes":1730},{"text":1729,"level":240},"Glossary",{},{"id":965,"data":1732,"type":965,"tunes":1753},{"title":1733,"entries":1734},"Core terms",[1735,1738,1740,1742,1745,1747,1749,1751],{"term":1736,"anchor":971,"definition":1737},"Generative model","An AI model designed to generate derived synthetic content such as text, images, audio, video, code or structured output.",{"term":1204,"anchor":327,"definition":1739},"The process of selecting relevant information from an external source or store for the current task.",{"term":756,"anchor":563,"definition":1741},"Retrieval-Augmented Generation: a pattern in which retrieved external information is supplied to a generative model to improve the current output.",{"term":1743,"anchor":567,"definition":1744},"Tool","A capability exposed to an AI runtime for reading data, calculating, searching, or performing an external action.",{"term":1210,"anchor":335,"definition":1746},"The information available to the model for a particular inference step.",{"term":1213,"anchor":983,"definition":1748},"The software layer that coordinates model calls, tool calls, task loops, sessions, retries, events or execution environments.",{"term":1216,"anchor":343,"definition":1750},"The product and domain layer that owns user interaction, authoritative state, permissions, validation, persistence and business behavior.",{"term":1445,"anchor":989,"definition":1752},"The service or runtime that exposes access to one or more models; provider identity and model identity are separate concerns.",{},{"id":993,"data":1755,"type":42,"tunes":1757},{"text":1756,"level":240},"Primary sources and implementation evidence",{},{"id":998,"data":1759,"type":218,"tunes":1761},{"text":1760},"Stable definitions below are anchored in standards\u002Fresearch; fast-moving implementation examples use current official engineering documentation. Aaasaasa AI Client is original implementation evidence and was checked against its codebase\u002Fdocumentation state dated 26 July 2026.",{},{"id":1003,"data":1763,"type":1010,"tunes":1768},{"link":1005,"meta":1764},{"image":1765,"title":1766,"description":1767},{"url":325},"NIST AI 600-1 — Generative Artificial Intelligence Profile","NIST's Generative AI profile, including the generative-AI definition and explicit distinction between model-, system-, application- and use-case-level concerns.",{},{"id":1013,"data":1770,"type":1010,"tunes":1775},{"link":1015,"meta":1771},{"image":1772,"title":1773,"description":1774},{"url":325},"NIST — Artificial Intelligence Model","Current NIST glossary definition of an AI model as a component of an information system that produces outputs from inputs using AI techniques.",{},{"id":1022,"data":1777,"type":1010,"tunes":1782},{"link":1024,"meta":1778},{"image":1779,"title":1780,"description":1781},{"url":325},"NIST — Artificial Intelligence System","Current NIST glossary definition showing that an AI system can include data systems, software, hardware, applications, tools or utilities using AI.",{},{"id":1031,"data":1784,"type":1010,"tunes":1789},{"link":1033,"meta":1785},{"image":1786,"title":1787,"description":1788},{"url":325},"Lewis et al. — Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","The 2020 paper introducing the RAG formulation that combines a generative model with retrieved non-parametric memory.",{},{"id":1040,"data":1791,"type":1010,"tunes":1796},{"link":1042,"meta":1792},{"image":1793,"title":1794,"description":1795},{"url":325},"OpenAI — File Search","Current official documentation for hosted file retrieval in the Responses API using uploaded-file knowledge bases, semantic search and keyword search.",{},{"id":1049,"data":1798,"type":1010,"tunes":1803},{"link":1051,"meta":1799},{"image":1800,"title":1801,"description":1802},{"url":325},"OpenAI — Function Calling","Current official documentation describing tool\u002Ffunction calling as the interface between models and external systems, data and actions.",{},{"id":1058,"data":1805,"type":1010,"tunes":1810},{"link":1060,"meta":1806},{"image":1807,"title":1808,"description":1809},{"url":325},"Anthropic — Effective Context Engineering for AI Agents","Engineering guidance defining context as the token set available during LLM sampling and explaining why context selection is a finite-resource problem.",{},"2.31.6","Generative AI is more than a model. Learn how models, retrieval, tools, context, runtimes and applications fit together in production AI systems.",{"lang":7,"title":208,"content":210,"contentJson":1814,"excerpt":1067},{"time":212,"blocks":1815,"version":1066},[1816,1819,1822,1825,1828,1831,1834,1837,1840,1843,1846,1858,1861,1864,1884,1887,1890,1893,1896,1899,1902,1905,1908,1911,1914,1917,1920,1923,1926,1929,1932,1935,1938,1941,1944,1947,1950,1953,1956,1959,1962,1965,1968,1980,1983,1986,2003,2006,2009,2012,2015,2018,2021,2033,2036,2039,2051,2054,2057,2077,2080,2083,2094,2097,2100,2103,2114,2117,2120,2123,2126,2129,2132,2135,2138,2141,2144,2147,2150,2153,2156,2159,2162,2165,2168,2171,2174,2185,2188,2200,2203,2206,2211,2216,2221,2226,2231,2236],{"id":215,"data":1817,"type":218,"tunes":1818},{"text":217},{},{"id":221,"data":1820,"type":226,"tunes":1821},{"body":223,"title":224,"variant":225},{},{"id":229,"data":1823,"type":226,"tunes":1824},{"body":231,"title":232,"variant":233},{},{"id":236,"data":1826,"type":241,"tunes":1827},{"title":238,"maxLevel":239,"minLevel":240},{},{"id":244,"data":1829,"type":42,"tunes":1830},{"text":246,"level":240},{},{"id":249,"data":1832,"type":218,"tunes":1833},{"text":251},{},{"id":254,"data":1835,"type":218,"tunes":1836},{"text":256},{},{"id":259,"data":1838,"type":226,"tunes":1839},{"body":261,"title":262,"variant":263},{},{"id":266,"data":1841,"type":42,"tunes":1842},{"text":268,"level":240},{},{"id":271,"data":1844,"type":218,"tunes":1845},{"text":273},{},{"id":276,"data":1847,"type":305,"tunes":1857},{"steps":1848,"title":303,"orientation":304},[1849,1850,1851,1852,1853,1854,1855,1856],{"label":280,"description":281},{"label":283,"description":284},{"label":286,"description":287},{"label":289,"description":290},{"label":292,"description":293},{"label":295,"description":296},{"label":298,"description":299},{"label":301,"description":302},{},{"id":308,"data":1859,"type":218,"tunes":1860},{"text":310},{},{"id":313,"data":1862,"type":42,"tunes":1863},{"text":315,"level":240},{},{"id":318,"data":1865,"type":358,"tunes":1883},{"rows":1866,"title":346,"layout":347,"columns":1879},[1867,1869,1871,1873,1875,1877],{"id":322,"label":323,"values":1868},[325,325,325],{"id":327,"label":328,"values":1870},[325,325,325],{"id":331,"label":332,"values":1872},[325,325,325],{"id":335,"label":336,"values":1874},[325,325,325],{"id":339,"label":340,"values":1876},[325,325,325],{"id":343,"label":344,"values":1878},[325,325,325],[1880,1881,1882],{"id":350,"label":351},{"id":353,"label":354},{"id":356,"label":357},{},{"id":361,"data":1885,"type":42,"tunes":1886},{"text":363,"level":240},{},{"id":366,"data":1888,"type":218,"tunes":1889},{"text":368},{},{"id":371,"data":1891,"type":218,"tunes":1892},{"text":373},{},{"id":376,"data":1894,"type":218,"tunes":1895},{"text":378},{},{"id":381,"data":1897,"type":42,"tunes":1898},{"text":383,"level":240},{},{"id":386,"data":1900,"type":218,"tunes":1901},{"text":388},{},{"id":391,"data":1903,"type":218,"tunes":1904},{"text":393},{},{"id":396,"data":1906,"type":226,"tunes":1907},{"body":398,"title":399,"variant":400},{},{"id":403,"data":1909,"type":409,"tunes":1910},{"url":405,"title":406,"excerpt":407,"ctaLabel":408},{},{"id":412,"data":1912,"type":42,"tunes":1913},{"text":414,"level":240},{},{"id":417,"data":1915,"type":218,"tunes":1916},{"text":419},{},{"id":422,"data":1918,"type":218,"tunes":1919},{"text":424},{},{"id":427,"data":1921,"type":218,"tunes":1922},{"text":429},{},{"id":432,"data":1924,"type":226,"tunes":1925},{"body":434,"title":435,"variant":263},{},{"id":438,"data":1927,"type":42,"tunes":1928},{"text":440,"level":240},{},{"id":443,"data":1930,"type":218,"tunes":1931},{"text":445},{},{"id":448,"data":1933,"type":218,"tunes":1934},{"text":450},{},{"id":453,"data":1936,"type":218,"tunes":1937},{"text":455},{},{"id":458,"data":1939,"type":42,"tunes":1940},{"text":460,"level":240},{},{"id":463,"data":1942,"type":218,"tunes":1943},{"text":465},{},{"id":468,"data":1945,"type":218,"tunes":1946},{"text":470},{},{"id":473,"data":1948,"type":218,"tunes":1949},{"text":475},{},{"id":478,"data":1951,"type":42,"tunes":1952},{"text":480,"level":240},{},{"id":483,"data":1954,"type":218,"tunes":1955},{"text":485},{},{"id":488,"data":1957,"type":218,"tunes":1958},{"text":490},{},{"id":493,"data":1960,"type":226,"tunes":1961},{"body":495,"title":496,"variant":225},{},{"id":499,"data":1963,"type":42,"tunes":1964},{"text":501,"level":240},{},{"id":504,"data":1966,"type":218,"tunes":1967},{"text":506},{},{"id":509,"data":1969,"type":347,"tunes":1979},{"content":1970,"stretched":43,"withHeadings":14},[1971,1972,1973,1974,1975,1976,1977,1978],[513,514,515],[517,518,519],[521,522,523],[525,526,527],[529,530,531],[533,534,535],[537,323,538],[540,541,542],{},{"id":545,"data":1981,"type":218,"tunes":1982},{"text":547},{},{"id":550,"data":1984,"type":42,"tunes":1985},{"text":552,"level":240},{},{"id":555,"data":1987,"type":358,"tunes":2002},{"rows":1988,"title":574,"layout":347,"columns":1997},[1989,1991,1993,1995],{"id":559,"label":560,"values":1990},[325,325,325,325],{"id":563,"label":564,"values":1992},[325,325,325,325],{"id":567,"label":568,"values":1994},[325,325,325,325],{"id":571,"label":572,"values":1996},[325,325,325,325],[1998,1999,2000,2001],{"id":327,"label":518},{"id":331,"label":332},{"id":579,"label":580},{"id":582,"label":583},{},{"id":586,"data":2004,"type":218,"tunes":2005},{"text":588},{},{"id":591,"data":2007,"type":42,"tunes":2008},{"text":593,"level":240},{},{"id":596,"data":2010,"type":226,"tunes":2011},{"body":598,"title":599,"variant":233},{},{"id":602,"data":2013,"type":218,"tunes":2014},{"text":604},{},{"id":607,"data":2016,"type":218,"tunes":2017},{"text":609},{},{"id":612,"data":2019,"type":218,"tunes":2020},{"text":614},{},{"id":617,"data":2022,"type":347,"tunes":2032},{"content":2023,"stretched":43,"withHeadings":14},[2024,2025,2026,2027,2028,2029,2030,2031],[621,622],[323,624],[626,627],[629,630],[632,633],[635,636],[638,639],[344,641],{},{"id":644,"data":2034,"type":226,"tunes":2035},{"body":646,"title":647,"variant":263},{},{"id":650,"data":2037,"type":42,"tunes":2038},{"text":652,"level":240},{},{"id":655,"data":2040,"type":347,"tunes":2050},{"content":2041,"stretched":43,"withHeadings":14},[2042,2043,2044,2045,2046,2047,2048,2049],[659,660],[662,663],[665,666],[668,669],[671,672],[674,675],[677,678],[680,681],{},{"id":684,"data":2052,"type":42,"tunes":2053},{"text":686,"level":240},{},{"id":689,"data":2055,"type":218,"tunes":2056},{"text":691},{},{"id":694,"data":2058,"type":358,"tunes":2076},{"rows":2059,"title":721,"layout":347,"columns":2072},[2060,2062,2064,2066,2068,2070],{"id":698,"label":699,"values":2061},[325,325,325],{"id":702,"label":703,"values":2063},[325,325,325],{"id":706,"label":707,"values":2065},[325,325,325],{"id":710,"label":711,"values":2067},[325,325,325],{"id":714,"label":715,"values":2069},[325,325,325],{"id":718,"label":719,"values":2071},[325,325,325],[2073,2074,2075],{"id":724,"label":725},{"id":727,"label":728},{"id":730,"label":731},{},{"id":734,"data":2078,"type":42,"tunes":2079},{"text":736,"level":240},{},{"id":739,"data":2081,"type":218,"tunes":2082},{"text":741},{},{"id":744,"data":2084,"type":347,"tunes":2093},{"content":2085,"stretched":43,"withHeadings":14},[2086,2087,2088,2089,2090,2091,2092],[748,749,750],[752,753,754],[756,757,758],[760,761,762],[764,765,766],[768,769,770],[772,773,774],{},{"id":777,"data":2095,"type":218,"tunes":2096},{"text":779},{},{"id":782,"data":2098,"type":42,"tunes":2099},{"text":784,"level":240},{},{"id":787,"data":2101,"type":218,"tunes":2102},{"text":789},{},{"id":792,"data":2104,"type":305,"tunes":2113},{"steps":2105,"title":816,"orientation":304},[2106,2107,2108,2109,2110,2111,2112],{"label":796,"description":797},{"label":799,"description":800},{"label":802,"description":803},{"label":805,"description":806},{"label":808,"description":809},{"label":811,"description":812},{"label":814,"description":815},{},{"id":819,"data":2115,"type":42,"tunes":2116},{"text":821,"level":240},{},{"id":824,"data":2118,"type":218,"tunes":2119},{"text":826},{},{"id":829,"data":2121,"type":218,"tunes":2122},{"text":831},{},{"id":834,"data":2124,"type":226,"tunes":2125},{"body":836,"title":837,"variant":263},{},{"id":840,"data":2127,"type":42,"tunes":2128},{"text":842,"level":240},{},{"id":845,"data":2130,"type":218,"tunes":2131},{"text":847},{},{"id":850,"data":2133,"type":409,"tunes":2134},{"url":852,"title":853,"excerpt":854,"ctaLabel":855},{},{"id":858,"data":2136,"type":409,"tunes":2137},{"url":860,"title":861,"excerpt":862,"ctaLabel":863},{},{"id":866,"data":2139,"type":42,"tunes":2140},{"text":868,"level":240},{},{"id":871,"data":2142,"type":218,"tunes":2143},{"text":873},{},{"id":876,"data":2145,"type":218,"tunes":2146},{"text":878},{},{"id":881,"data":2148,"type":218,"tunes":2149},{"text":883},{},{"id":886,"data":2151,"type":42,"tunes":2152},{"text":888,"level":240},{},{"id":891,"data":2154,"type":218,"tunes":2155},{"text":893},{},{"id":896,"data":2157,"type":218,"tunes":2158},{"text":898},{},{"id":901,"data":2160,"type":42,"tunes":2161},{"text":903,"level":240},{},{"id":906,"data":2163,"type":218,"tunes":2164},{"text":908},{},{"id":911,"data":2166,"type":218,"tunes":2167},{"text":913},{},{"id":916,"data":2169,"type":218,"tunes":2170},{"text":918},{},{"id":921,"data":2172,"type":42,"tunes":2173},{"text":923,"level":240},{},{"id":926,"data":2175,"type":926,"tunes":2184},{"items":2176,"title":957},[2177,2178,2179,2180,2181,2182,2183],{"id":930,"answer":931,"question":932},{"id":934,"answer":935,"question":936},{"id":938,"answer":939,"question":940},{"id":942,"answer":943,"question":944},{"id":946,"answer":947,"question":948},{"id":950,"answer":951,"question":952},{"id":954,"answer":955,"question":956},{},{"id":960,"data":2186,"type":42,"tunes":2187},{"text":962,"level":240},{},{"id":965,"data":2189,"type":965,"tunes":2199},{"title":967,"entries":2190},[2191,2192,2193,2194,2195,2196,2197,2198],{"term":970,"anchor":971,"definition":972},{"term":518,"anchor":327,"definition":974},{"term":756,"anchor":563,"definition":976},{"term":978,"anchor":567,"definition":979},{"term":336,"anchor":335,"definition":981},{"term":340,"anchor":983,"definition":984},{"term":344,"anchor":343,"definition":986},{"term":988,"anchor":989,"definition":990},{},{"id":993,"data":2201,"type":42,"tunes":2202},{"text":995,"level":240},{},{"id":998,"data":2204,"type":218,"tunes":2205},{"text":1000},{},{"id":1003,"data":2207,"type":1010,"tunes":2210},{"link":1005,"meta":2208},{"image":2209,"title":1008,"description":1009},{"url":325},{},{"id":1013,"data":2212,"type":1010,"tunes":2215},{"link":1015,"meta":2213},{"image":2214,"title":1018,"description":1019},{"url":325},{},{"id":1022,"data":2217,"type":1010,"tunes":2220},{"link":1024,"meta":2218},{"image":2219,"title":1027,"description":1028},{"url":325},{},{"id":1031,"data":2222,"type":1010,"tunes":2225},{"link":1033,"meta":2223},{"image":2224,"title":1036,"description":1037},{"url":325},{},{"id":1040,"data":2227,"type":1010,"tunes":2230},{"link":1042,"meta":2228},{"image":2229,"title":1045,"description":1046},{"url":325},{},{"id":1049,"data":2232,"type":1010,"tunes":2235},{"link":1051,"meta":2233},{"image":2234,"title":1054,"description":1055},{"url":325},{},{"id":1058,"data":2237,"type":1010,"tunes":2240},{"link":1060,"meta":2238},{"image":2239,"title":1063,"description":1064},{"url":325},{},"Post erfolgreich abgerufen",{"items":2243,"source":2327,"manualIds":2328,"manualMatchedIds":2329},[2244,2251,2258,2265,2272,2279,2286,2293,2300,2307,2314,2321],{"id":2245,"slug":2246,"title":2247,"excerpt":2248,"featuredImage":2249,"publishedAt":2250},"485","enterprise-ai-architecture-what-changes-when-ai-enters-a-company","Архитектура ИИ на предприятии: что меняется, когда ИИ приходит в компанию","Архитектура ИИ для предприятий объясняет, как ИИ меняет корпоративные системы в таких областях, как полномочия на данные, идентификация, разрешения, поставщики, риски, управление, оценка, соответствие требованиям и операции.","\u002Fuploads\u002F2026\u002F10\u002Fenterprise-ai-architecture-what-changes-when-ai-enters-a-company-1791478161363-czrwaq.webp","2026-10-08T10:48:00.000Z",{"id":2252,"slug":2253,"title":2254,"excerpt":2255,"featuredImage":2256,"publishedAt":2257},"486","source-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from","Источник истины в системах ИИ: откуда на самом деле берутся надёжные знания","Источник истины определяет, какой источник является авторитетным для конкретного факта или состояния. Узнайте, чем он отличается от RAG, происхождения данных, памяти, контекста, векторных баз данных и систем учёта.","\u002Fuploads\u002F2026\u002F10\u002Fsource-of-truth-in-ai-systems-where-reliable-knowledge-actually-comes-from-1791479103235-6bq9em.webp","2026-10-08T13:02:00.000Z",{"id":2259,"slug":2260,"title":2261,"excerpt":2262,"featuredImage":2263,"publishedAt":2264},"489","agentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act","Агентный ИИ: когда система ИИ может планировать, использовать инструменты и действовать","Агентный ИИ использует модели внутри многошаговых циклов выполнения, где они могут выбирать инструменты, наблюдать результаты, обновлять состояние и адаптировать своё следующее действие в рамках явных границ времени выполнения и разрешений.","\u002Fuploads\u002F2026\u002F10\u002Fagentic-ai-explained-when-an-ai-system-can-plan-use-tools-and-act-1791481499084-wnji2a.webp","2026-10-08T11:43:00.000Z",{"id":2266,"slug":2267,"title":2268,"excerpt":2269,"featuredImage":2270,"publishedAt":2271},"484","what-is-an-ai-platform-architect-models-data-runtime-security-and-operations","Что такое архитектор AI-платформы? Модели, данные, среда выполнения, безопасность и операции","Архитектор платформы ИИ проектирует многоразовые основы ИИ для моделей, провайдеров, поиска, агентов, идентификации, безопасности, оценки, наблюдаемости и операций.","\u002Fuploads\u002F2026\u002F10\u002Fwhat-is-an-ai-platform-architect-models-data-runtime-security-and-operations-1791477229171-ou3zcc.webp","2026-10-08T12:32:00.000Z",{"id":2273,"slug":2274,"title":2275,"excerpt":2276,"featuredImage":2277,"publishedAt":2278},"487","vector-databases-embeddings-and-reranking-three-different-parts-of-retrieval","Векторные базы данных, эмбеддинги и переранжирование: три разные части поиска","Эмбеддинги представляют смысл, векторные базы данных извлекают кандидатов, а реранкеры уточняют результаты. Узнайте, чем отличаются эти три слоя поиска и как они работают вместе в RAG.","\u002Fuploads\u002F2026\u002F10\u002Fvector-databases-embeddings-and-reranking-three-different-parts-of-retrieval-1791480129884-9dtasz.webp","2026-10-08T11:21:00.000Z",{"id":2280,"slug":2281,"title":2282,"excerpt":2283,"featuredImage":2284,"publishedAt":2285},"470","what-should-an-ai-agent-remember-forget-recompute-or-retrieve-again","Что ИИ-агент должен помнить, забывать, перевычислять или извлекать повторно?","Долгоживущие агенты не должны помнить всё. В этой статье представлена практическая модель жизненного цикла для определения того, что относится к долговременной памяти, что следует извлекать повторно, что безопаснее пересчитать, а что должно истечь по сроку действия или быть заменено.","\u002Fuploads\u002F2026\u002F09\u002Fwhat-should-an-ai-agent-remember-forget-recompute-or-retrieve-again-1790351131087-iehz28.webp","2026-09-25T09:43:00.000Z",{"id":2287,"slug":2288,"title":2289,"excerpt":2290,"featuredImage":2291,"publishedAt":2292},"467","the-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers","Граница достоверности ответа: недостающий слой между релевантностью и надёжными ответами ИИ","Источник может быть релевантным, авторитетным и при этом неверным для задаваемого вопроса. Недостающий слой — применимость: условия, при которых ответ остаётся в силе, и изменения, вынуждающие пересмотреть его. В этой статье вводится понятие «Граница действительности ответа» как паттерн проектирования источников для людей, ИИ-поиска и RAG-систем.","\u002Fuploads\u002F2026\u002F09\u002Fthe-answer-validity-boundary-the-missing-layer-between-relevance-and-reliable-ai-answers-1790272901306-1g5jly.webp","2026-09-24T11:59:00.000Z",{"id":2294,"slug":2295,"title":2296,"excerpt":2297,"featuredImage":2298,"publishedAt":2299},"490","rbac-vs-tenant-isolation-two-different-security-boundaries","RBAC против изоляции арендаторов: две разные границы безопасности","RBAC определяет, что пользователь может делать; изоляция тенантов определяет, к ресурсам какого тенанта это действие может получить доступ. Узнайте, почему безопасность многотенантного SaaS требует обеих границ.","\u002Fuploads\u002F2026\u002F10\u002Frbac-vs-tenant-isolation-two-different-security-boundaries-1791485111528-qqtzby.webp","2026-10-08T14:43:00.000Z",{"id":2301,"slug":2302,"title":2303,"excerpt":2304,"featuredImage":2305,"publishedAt":2306},"495","sovereign-ai-control-of-models-data-infrastructure-and-dependencies","Суверенный ИИ: контроль над моделями, данными, инфраструктурой и зависимостями","Суверенный ИИ — это эффективный контроль над моделями, данными, инфраструктурой, программным обеспечением, операциями и стратегическими зависимостями, а не просто место размещения модели ИИ.","\u002Fuploads\u002F2026\u002F10\u002Fsovereign-ai-control-of-models-data-infrastructure-and-dependencies-1791488833132-niy85x.webp","2026-10-08T15:45:00.000Z",{"id":2308,"slug":2309,"title":2310,"excerpt":2311,"featuredImage":2312,"publishedAt":2313},"476","mcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained","MCP vs A2A vs UCP vs AP2 vs A2UI: разбор стека протоколов агентов","MCP, A2A, UCP, AP2 и A2UI часто представляют как конкурирующие агентские стандарты. В основном они решают разные проблемы интероперабельности. Это руководство сопоставляет каждый протокол с границей, которую он фактически стандартизирует,—и показывает, как они могут работать вместе в одной промышленной системе.","\u002Fuploads\u002F2026\u002F09\u002Fmcp-vs-a2a-vs-ucp-vs-ap2-vs-a2ui-the-agent-protocol-stack-explained-1790352625869-2ezle0.webp","2026-09-25T12:09:00.000Z",{"id":2315,"slug":2316,"title":2317,"excerpt":2318,"featuredImage":2319,"publishedAt":2320},"482","adr-vs-nfr-architecture-decisions-and-system-quality-are-not-the-same-thing","ADR vs NFR: архитектурные решения и качество системы — это не одно и то же","ADR против NFR: узнайте, как требования к качеству системы определяют архитектурные решения, как ADR фиксируют компромиссы и почему валидация остаётся отдельной.","\u002Fuploads\u002F2026\u002F10\u002Fadr-vs-nfr-architecture-decisions-and-system-quality-are-not-the-same-thing-1791475921511-6zgen1.webp","2026-10-08T12:11:00.000Z",{"id":2322,"slug":2323,"title":861,"excerpt":2324,"featuredImage":2325,"publishedAt":2326},"480","when-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger","Модель ИИ не нуждается в поиске для каждого вопроса. Важная проблема — знать, когда её внутренних знаний уже недостаточно. Триггер поиска — это практическая граница принятия решений, которая определяет, когда система ИИ должна перестать полагаться исключительно на знания модели и получить внешние доказательства перед ответом.","\u002Fuploads\u002F2026\u002F09\u002Fwhen-should-an-ai-stop-trusting-its-own-knowledge-the-retrieval-trigger-1790574991244-f4rpyg.webp","2026-09-28T01:49:00.000Z","fallback",[],[]]