AISE502/Folien/AISE502_Vorlesung_12_Folien.tex

497 lines
36 KiB
TeX

% !TEX encoding = UTF-8 Unicode
% ============================================================================
% AISE502 -- AI in Software Engineering II
% Lecture 12 slides, typeset with the official FHGR beamer theme
% (beamerthemeFHGR.sty, University of Applied Sciences of the Grisons).
% Slide content is unchanged; only the presentation layer is the FHGR template.
% ============================================================================
\documentclass[aspectratio=169]{beamer}
\usetheme[showsection, titlebg=pics/theme_pics/titlepage.png]{FHGR}
% ============================================
% PACKAGES
% The theme already loads tikz, graphicx, xcolor, tabularx, colortbl,
% listings, hyperref, environ and xparse -- only the extras are needed here.
% ============================================
\usepackage[british]{babel}
\usepackage{booktabs}
\usepackage{amsmath}
\usepackage{amssymb}
\usepackage{tcolorbox}
\usetikzlibrary{shapes.geometric, arrows.meta, positioning, fit, backgrounds, calc}
% ============================================
% SEMANTIC COLOURS, MAPPED ONTO THE FHGR PALETTE
% The names used throughout the slides are kept, so no slide text changes;
% they now resolve to the FHGR brand colours defined by the theme.
% ============================================
\colorlet{bankblue}{blue} % FHGR blue (4B92A4)
\colorlet{bankgreen}{green} % FHGR green (817E65)
\colorlet{bankred}{red} % FHGR red (C60219)
\colorlet{codegray}{gray} % FHGR gray (595959)
\colorlet{backcolour}{linen} % FHGR linen (E1D3B5)
\definecolor{aiviolet}{HTML}{6B4E71} % muted plum, kept distinct for the AI lens
% Attribution labels in English (theme default is German)
\renewcommand{\source}[1]{\par\hfill {\tiny\color{FHGRDeco} Source:\,\itshape #1}}
\renewcommand{\imagesource}[1]{\par\hfill {\tiny\color{FHGRDeco} Image source:\,\itshape #1}}
% ============================================
% CUSTOM TCOLORBOXES (same semantics as the script, FHGR colours)
% ============================================
\newtcolorbox{keypoint}{
colback=bankblue!7!white,
colframe=bankblue,
title=Key Concept,
fonttitle=\bfseries\small,
boxrule=0.8pt,
arc=2pt,
top=2pt, bottom=2pt, left=4pt, right=4pt
}
\newtcolorbox{examplebox}[1][]{
colback=bankgreen!10!white,
colframe=bankgreen,
title={Example: #1},
fonttitle=\bfseries\small,
boxrule=0.8pt,
arc=2pt,
top=2pt, bottom=2pt, left=4pt, right=4pt
}
\newtcolorbox{definitionbox}[1][]{
colback=linen!40!white,
colframe=camel!85!black,
title={Definition: #1},
fonttitle=\bfseries\small,
boxrule=0.8pt,
arc=2pt,
top=2pt, bottom=2pt, left=4pt, right=4pt
}
\newtcolorbox{thinkbox}{
colback=lightGray!35!white,
colframe=darkGray,
title=Discussion,
fonttitle=\bfseries\small,
boxrule=0.8pt,
arc=2pt,
top=2pt, bottom=2pt, left=4pt, right=4pt
}
\newtcolorbox{hinweisbox}{
colback=bankred!5!white,
colframe=bankred,
title=Important Note,
fonttitle=\bfseries\small,
boxrule=0.8pt,
arc=2pt,
top=2pt, bottom=2pt, left=4pt, right=4pt
}
\newtcolorbox{ailinse}[1][]{
colback=aiviolet!7!white,
colframe=aiviolet,
title={AI Lens: #1},
fonttitle=\bfseries\small,
boxrule=0.8pt,
arc=2pt,
top=2pt, bottom=2pt, left=4pt, right=4pt
}
\newtcolorbox{projektbox}{
colback=bankblue!4!white,
colframe=bankblue!70!black,
title=Project Link: Portfolio Intelligence Platform,
fonttitle=\bfseries\small,
boxrule=0.8pt,
arc=2pt,
top=2pt, bottom=2pt, left=4pt, right=4pt
}
% ============================================
% TITLE METADATA
% ============================================
\title[AI in Software Engineering II]{AISE502: AI in Software Engineering II}
\subtitle{Lecture 12: The AI Dimension I -- Axis A Evidence, Axis B Foundations\\[0.4ex]{\small Script: Part V, Sections 40--41, 42.1--42.5}}
\author{Dr.\ Florian Herzog}
\shortname{AISE502}
\fullname{Fachhochschule Graub\"unden, Chur -- Autumn Semester 2026}
\begin{document}
% ============================================
% TITLE SLIDE
% ============================================
\FHGRTitlePage
% ============================================
% AGENDA
% ============================================
\begin{frame}{Agenda}
\small
\begin{enumerate}\setlength\itemsep{1pt}
\item \textbf{Two axes, one method} -- Assumption A6 falls due
\item \textbf{Axis A}: two contradictory RCTs and the empirical record
\item Reconciling the divergence; the verification bottleneck (\textbf{Maxim 7}); Axis A compact
\item \textbf{Axis B}: the news-sentiment call, wired the obvious way
\item The three component types; why containment: the SE4AI classics
\item The reference architecture: \textbf{LLM gateway, queue, ontology guard}
\item The eval harness as an engineering artefact
\item This week's exercise: \textbf{AdvisorAgent $+$ sub-agents behind the gateway}
\end{enumerate}
\end{frame}
% ============================================
% RECAP
% ============================================
\section{Recap}
\begin{frame}{Recap: where we are}
\footnotesize
\begin{itemize}\setlength\itemsep{3pt}
\item \textbf{Parts I--IV complete; Lecture 11 closed Part IV}: fitness functions -- three families (dependency gates, budgets, chaos experiments); the four-layer cascade and the eight-row C10 reference contract; cost of change flat \emph{within} / steep \emph{across} architecture boundaries; Conway -- the fit is three-way; six limits; \textbf{Maxim 9} -- Part IV closed
\item \textbf{AI Lens threads so far} -- deck 1: the two AI axes and Assumption A6; deck 3: an LLM component stresses D3, D10, D12; an agent drafts the ADR, a human owns the decision; \textbf{ADR-011}: all LLM calls through one gateway port
\item \textbf{Deck 6}: C10 profile -- D12 $=$ H, evals as the operative meaning of testability, cost per request; outlook: agent orchestration reuses the catalogue's topologies, workflows before agents ($15\times$ token finding); \textbf{deck 11}: fitness functions are the operating licence for agents (A); eval pass rate and token budget are fitness functions with old mechanics (B)
\item \textbf{Today}: Part V redeems A6 systematically -- the two axes as one method; the Axis A evidence and its resolution; the Axis B foundations up to the eval harness
\item \textbf{Project}: M4 delivered (deterministic core fully tested and resilient); \textbf{M5 begins} -- AdvisorAgent $+$ 2--3 sub-agents behind the gateway, ontology guard active; the reference architecture today is just-in-time
\end{itemize}
\end{frame}
% ============================================
% TWO AXES, ONE METHOD
% ============================================
\section{Two Axes, One Method}
\begin{frame}{Part V opens: the promissory note falls due}
\emph{\textcolor{bankblue}{Four parts built a complete decision theory without ever making artificial intelligence its subject -- does the construction survive the technology that defines its decade?}}
\vspace{0.2cm}
\small
\begin{itemize}\setlength\itemsep{4pt}
\item Not a rhetorical flourish but a \textbf{promissory note falling due}: Part I issued it as \textbf{Assumption A6} -- \emph{AI components extend the quality attribute space but do not change the method}. The bet in two sentences: everything AI does to software engineering can be absorbed by the apparatus you now own
\item If AI-bearing systems required a genuinely different method, the bet would be lost -- this part is where the claim must \textbf{survive contact with the evidence}
\item Roadmap -- \textbf{cases first, generalisation after}: two contradictory randomised experiments open Axis A (\S41); one concrete LLM call, wired wrongly and then rightly, opens Axis B (\S42); the matrix reading (\S43) and the emergent pattern (\S44) follow next week
\end{itemize}
\end{frame}
\begin{frame}{The two axes of the AI dimension}
\begin{columns}[T]
\begin{column}{0.55\textwidth}
\begin{definitionbox}[The two axes of the AI dimension]
\footnotesize
\begin{itemize}\setlength\itemsep{2pt}
\item \textbf{Axis A -- AI as a tool in the SDLC.} Code assistants, agentic coding tools, review bots participate in \emph{building} the software (code, tests, documentation, design drafts); what ships may contain no AI at all. Unit of analysis: the \emph{development process} and its economics
\item \textbf{Axis B -- AI as a runtime component.} LLM services, trained ML models, optimisation solvers are \emph{part of the delivered system} and execute in production. Unit of analysis: the \emph{running system} and its quality attributes
\item \textbf{The axes are independent}: a payroll system built with heavy agent support (A without B); a hand-crafted AI-native advisory platform (B without A). In the course project both apply simultaneously -- which is why they must be kept conceptually apart
\end{itemize}
\end{definitionbox}
\end{column}
\begin{column}{0.42\textwidth}
\begin{center}
\resizebox{\linewidth}{!}{%
\begin{tikzpicture}[
sysbox/.style={rectangle, draw, rounded corners=4pt, minimum width=3.6cm, minimum height=1.0cm, align=center, font=\small\sffamily, line width=0.8pt},
ax/.style={sysbox, fill=aiviolet!15, draw=aiviolet},
proc/.style={sysbox, fill=bankgreen!15, draw=bankgreen},
core/.style={sysbox, fill=bankblue!20, draw=bankblue, font=\small\sffamily\bfseries},
arr/.style={-{Stealth[length=2.5mm]}, thick, gray!60!black}
]
\node[ax] (axisa) at (0,2.4) {\textbf{Axis A}\\AI as tool: agents, assistants};
\node[ax] (axisb) at (5.0,2.4) {\textbf{Axis B}\\AI as component: LLM, ML, solver};
\node[proc] (sdlc) at (0,0) {Development process\\(specify, build, verify, operate)};
\node[core] (system) at (5.0,0) {Delivered system\\(structure, quality attributes)};
\draw[arr] (axisa) -- node[right, font=\scriptsize\sffamily, align=left]{shifts SDLC\\economics} (sdlc);
\draw[arr] (axisb) -- node[right, font=\scriptsize\sffamily, align=left]{stretches quality\\attribute space} (system);
\draw[arr] (sdlc) -- node[above, font=\scriptsize\sffamily]{produces} (system);
\end{tikzpicture}%
}
\end{center}
\vspace{-0.1cm}
\scriptsize Axis A changes \emph{how} systems are built; Axis B \emph{what} the built system contains -- both absorbed by the same method: scenarios, tactics, profiles, ADRs, fitness functions.
\vspace{0.1cm}
\begin{keypoint}
\footnotesize \textbf{Two axes, one method}: independent, kept apart -- and both analysed with the apparatus of Parts I--IV, \emph{nothing new}.
\end{keypoint}
\end{column}
\end{columns}
\end{frame}
% ============================================
% AXIS A -- EVIDENCE AND RESOLUTION
% ============================================
\section{Axis A -- Evidence and Resolution}
\begin{frame}{Case 1 -- the Copilot RCT: $+55.8\,\%$ on a greenfield task}
\emph{\textcolor{bankblue}{AI makes developers 55.8\,\% faster -- or 19\,\% slower. Which study is wrong?}}
\vspace{0.1cm}
\small Both numbers come from randomised controlled trials, both methodologically sound; few topics in software engineering carry a larger gap between headline and evidence.
\vspace{0.15cm}
\small
\begin{itemize}\setlength\itemsep{2pt}
\item \textbf{Case 1 (published 2023)}: 95 professional developers randomly split into two groups; same task -- implement an HTTP server in JavaScript; one group with GitHub Copilot, one without; the clock measured time to completion
\item The treatment group finished \textbf{55.8\,\% faster}
\item \textbf{Qualification 1}: the confidence interval (\textbf{21--89\,\%}) is very wide -- the headline number is a point estimate, not a natural constant
\item \textbf{Qualification 2}: a bounded, well-defined \emph{greenfield} exercise -- no legacy context, no architectural constraints, no review process
\item \textbf{Qualification 3}: speed was measured, not quality; completion rates did not differ significantly
\item Within those bounds the result is real -- and it is the origin of the ``AI doubles productivity'' headline genre
\end{itemize}
\end{frame}
\begin{frame}{Case 2 -- the METR RCT: $19\,\%$ slower in your own mature codebase}
\emph{\textcolor{bankblue}{What happens when the same technology meets experts on their own terrain?}}
\vspace{0.15cm}
\footnotesize
\begin{itemize}\setlength\itemsep{3pt}
\item \textbf{16 experienced open-source maintainers}; 246 real issues in repositories they had maintained for years -- large, mature codebases (over a million lines) with high implicit quality standards; each issue randomly assigned to an AI-allowed condition (predominantly Cursor with frontier models of early 2025) or an AI-forbidden condition
\item With AI, the developers took \textbf{19\,\% longer}
\item \textbf{The perception data are the didactic core}: forecast before the study \textbf{$+24\,\%$} speed-up; measured \textbf{$-19\,\%$}; post-hoc estimate \textbf{$+20\,\%$} faster -- even experts cannot validly introspect their own AI-assisted productivity
\item METR's own explanation \emph{maps boundary conditions} rather than refuting Case 1: deep repository familiarity left little for AI-supplied context to add; codebases large and conventionally dense; substantial time spent checking, repairing, discarding AI proposals
\end{itemize}
\end{frame}
\begin{frame}{The full empirical record, 2023--2025}
\footnotesize The two cases are the extreme corners of a larger record -- seven strands, 2023--2025, from randomised experiments to organisational telemetry and longitudinal code analysis; read every row \emph{setting first, finding second}.
\vspace{-0.1cm}
\scriptsize
\renewcommand{\arraystretch}{0.8}%
\begin{center}
\begin{tabular}{@{}p{2.1cm}p{4.3cm}p{7.0cm}@{}}
\toprule
\textbf{Evidence} & \textbf{Setting} & \textbf{Finding} \\
\midrule
Copilot RCT & 95 developers; greenfield HTTP server (JavaScript) & \textbf{$+55.8\,\%$} task speed (95\,\% CI 21--89\,\%); completion rate not significantly different \\
Three field experiments & 4{,}867 developers; Microsoft, Accenture, Fortune-100 firm & \textbf{$+26.1\,\%$} completed tasks (s.e.\ 10.3\,\%); less experienced developers gain most \\
METR RCT & 16 expert OSS maintainers; 246 real issues, own mature repositories & \textbf{19\,\% slower} with AI -- while estimating afterwards that AI had made them 20\,\% faster \\
DORA 2024 & $\sim$3{,}000 respondents; organisational delivery level & $+25\,\%$ AI adoption associated with \textbf{$-1.5\,\%$ throughput} and \textbf{$-7.2\,\%$ delivery stability} \\
DORA 2025 & $\sim$5{,}000 respondents & throughput association now positive; \textbf{instability persists}; AI acts as an \emph{amplifier} of existing strengths and dysfunctions \\
GitClear longitudinal & 211 million changed code lines, 2020--2024 & \textbf{$4\times$} growth in code duplication; moved-code share (the refactoring signature) collapsed from $\sim$25\,\% to below 10\,\% \\
Stack Overflow survey & $>$49{,}000 developers & 84\,\% use or plan to use AI; \textbf{46\,\% actively distrust} its output; top frustration: ``almost right'' code \\
\bottomrule
\end{tabular}
\end{center}
\vspace{-0.15cm}
\scriptsize \textcolor{codegray}{Caution from GitHub's own telemetry-plus-survey study: the best predictor of \emph{perceived} productivity is the suggestion acceptance rate, not the persistence of accepted code in the repository -- much vendor-reported ``productivity'' evidence measures perception, not verified output. Case 2's perception gap is the controlled-trial demonstration of the same fact.}
\end{frame}
\begin{frame}{The system level: DORA 2024 and 2025}
\footnotesize
\begin{itemize}\setlength\itemsep{1pt}
\item DORA measures neither task times nor perceptions but \textbf{delivery performance at the level of the organisation} -- throughput and stability -- exactly the level at which architecture acts
\item \textbf{2024} ($\sim$3{,}000 respondents; 75.9\,\% use AI for at least part of their work, roughly three quarters report productivity gains) -- a 25\,\% increase in AI adoption is associated with:
\begin{center}
\scriptsize
\renewcommand{\arraystretch}{0.85}%
\begin{tabular}{@{}ll@{}}
\toprule
\textbf{Gains} & \textbf{Losses} \\
\midrule
$+7.5\,\%$ documentation quality & $-1.5\,\%$ delivery throughput \\
$+3.4\,\%$ code quality & $-7.2\,\%$ delivery stability \\
$+3.1\,\%$ review speed & \\
\bottomrule
\end{tabular}
\end{center}
\vspace{-0.1cm}
Proposed mechanism is classical: more code per change, and larger batch sizes have been a documented risk driver for years
\item \textbf{2025} ($\sim$5{,}000 respondents): adoption near saturation (90\,\%, median about two hours of daily use); more than 80\,\% report productivity gains; 30\,\% still express little or no trust in AI-generated code; the throughput association has \textbf{turned positive} as tools and practices matured -- the negative association with delivery stability \textbf{persists}
\item Central metaphor: \textbf{AI is an amplifier} -- it magnifies the strengths of well-run organisations and the dysfunctions of badly run ones
\item \emph{Individual acceleration and system-level performance are different quantities, and only the second one pays salaries}
\end{itemize}
\end{frame}
\begin{frame}{Code structure and practitioner trust in the longitudinal record}
\footnotesize
\begin{itemize}\setlength\itemsep{4pt}
\item \textbf{GitClear}, 211 million changed lines (2020--2024): duplicated code blocks (five or more lines) at \textbf{four times} their pre-AI level in 2024; \emph{moved} lines -- the fingerprint of refactoring and modularisation -- fell from roughly \textbf{25\,\% to under 10\,\%}: 2024 was the first year in which copy-paste exceeded code movement
\item \textbf{Churn} -- code reworked or discarded within two weeks of commit -- rose from a pre-AI baseline of roughly 3--4\,\% to \textbf{5.7\,\%} in 2024, trend continuing; two obligatory caveats: GitClear is a commercial analytics vendor, and the analysis is \emph{correlational} -- AI's causal share is plausible but not isolated
\item Converges with DORA's stability data: \textbf{more code, produced faster, structurally worse maintained} -- reuse by abstraction displaced by reuse by duplication, the opposite of what Parnas-style modularisation (Part II) works to achieve
\item \textbf{Stack Overflow 2025} ($>$49{,}000 developers): \textbf{84\,\%} use or plan to use AI tools; \textbf{46\,\%} actively distrust the accuracy of the output; most-cited frustration (45\,\%): ``almost right, but not quite'' -- \emph{adoption rises while trust falls}, consistent with METR and DORA: the effort has migrated from writing to verifying
\end{itemize}
\end{frame}
\begin{frame}{Reconciling the divergence: five moderator variables}
\footnotesize \textbf{So which study is wrong? Neither} -- resolving the contradiction \emph{is} the lesson: different populations (task novices vs.\ domain experts in their own code), different codebases (greenfield vs.\ mature), different tasks (bounded vs.\ real issues) -- the results never actually compete; the resolution requires reading \emph{study designs}, not abstracts. Indexed by their moderator variables, Case 1 and Case 2 sit at opposite corners of a five-dimensional design space -- and every other row of the record finds its place in the same coordinates.
\vspace{0.1cm}
\footnotesize
\renewcommand{\arraystretch}{0.9}%
\begin{center}
\begin{tabular}{@{}p{2.2cm}p{5.0cm}p{5.2cm}@{}}
\toprule
\textbf{Moderator} & \textbf{Gains high} & \textbf{Gains low or negative} \\
\midrule
Experience & juniors, task novices (Copilot RCT, field experiments) & domain experts in their own code (METR) \\
Codebase & greenfield, small, standard stack & mature, large, dense implicit conventions \\
Task & well-defined, bounded & under-specified, cross-cutting \\
Measurement & task time, perceived productivity & delivery stability, maintainability, churn (DORA 2024, GitClear) \\
Organisation & small batches, test automation, loose coupling & large batches, weak guardrails, tight coupling (DORA 2025) \\
\bottomrule
\end{tabular}
\end{center}
\vspace{0.05cm}
\footnotesize The same technology yields $+55.8\,\%$ and $-19\,\%$ because the two cases differ on \textbf{every one of the five rows}.
\end{frame}
\begin{frame}{Discussion: which setting is yours?}
\begin{thinkbox}
\footnotesize
\begin{itemize}\setlength\itemsep{4pt}
\item The same class of technology produced $+55.8\,\%$ in one randomised experiment and $-19\,\%$ in another. Walk through the five moderators: \textbf{on which rows do the two studies differ?}
\item Consider the systems you are likely to work on two years after graduation -- greenfield exercises, or mature codebases with implicit conventions? \textbf{Which study's setting is closer to that reality?}
\item What does the METR perception gap (forecast $+24\,\%$, measured $-19\,\%$, post-hoc estimate $+20\,\%$) imply about \textbf{relying on your own felt productivity as evidence}?
\end{itemize}
\end{thinkbox}
\vspace{0.2cm}
\scriptsize \textcolor{codegray}{\textbf{Project transfer:} which moderator row describes your repository in week 12?}
\end{frame}
\begin{frame}{The verification bottleneck}
\small The structural conclusion underneath the moderator table, in one sentence:
\vspace{0.1cm}
\begin{center}
\textbf{Code generation became cheap; specification, verification, and architecture became the binding constraints.}
\end{center}
\vspace{0.1cm}
\small
\begin{itemize}\setlength\itemsep{4pt}
\item When the marginal cost of producing plausible code approaches zero, the scarce resource is no longer typing but \textbf{everything that surrounds it}: understanding the requirement precisely enough to specify it, reviewing and testing what was generated, and accepting responsibility for shipping it
\item \textbf{The strands converge}: DORA -- individual acceleration coexists with delivery instability where control systems are weak; two thirds of surveyed developers spend more time on almost-right code; a substantial share of METR's slow-down is time spent checking, repairing, discarding AI proposals; industry analyses describe \emph{code review as the new bottleneck} -- more and larger pull requests meeting unchanged human review capacity
\item Economically put: \textbf{AI lowers the cost of \emph{producing} code, not the cost of \emph{taking responsibility} for code}
\end{itemize}
\end{frame}
\begin{frame}{Three consequences bind Axis A into the fit theory -- Maxim 7}
\footnotesize
\begin{enumerate}\setlength\itemsep{2pt}
\item \textbf{Architecture quality gates AI gains} -- DORA 2025's core finding: teams in loosely coupled architectures with fast feedback loops convert AI adoption into throughput; tightly coupled systems with slow processes do not -- the AI-era echo of loosely coupled architectures and teams as the strongest predictor of continuous delivery performance \textcolor{codegray}{(the coupling finding of Lecture 11)}. In the theory's vocabulary: \textbf{D7} (evolvability) and \textbf{D9} (testability and deployability) gain weight in \emph{every} requirements profile -- architecture--application fit acquires a second reading: fit to a \emph{mode of work} in which change volume rises by an order of magnitude
\item \textbf{Architecture documentation becomes a control interface} -- ADRs, repository convention files, and machine-readable rules are no longer passive records; agents execute them on every run (\S41.6)
\item \textbf{Fitness functions become the operating licence for agents} -- an agent iterating against a dense test suite and CI-enforced architecture rules is contained; without them, every agent change is unpriced risk (\S41.7)
\end{enumerate}
\vspace{0.1cm}
\begin{keypoint}
\footnotesize \textbf{Maxim 7.} Good architecture was always the art of making change cheap and safe; AI raises the change rate by an order of magnitude -- and therefore \textbf{raises, not lowers, the value of architecture}.
\end{keypoint}
\end{frame}
% ============================================
% AXIS A -- CONTROL INTERFACE AND GUARDRAILS
% ============================================
\section{Axis A -- Control Interface and Guardrails}
\begin{frame}{Architecture documentation as a control interface for agents (1/2)}
\footnotesize
\begin{itemize}\setlength\itemsep{5pt}
\item Agentic tools are \textbf{context-driven}: they produce architecture-conformant code only if the architecture is \emph{explicit, machine-readable, and in the repository} -- Part I's documentation artefacts, written for human readers, upgrade into a \textbf{control interface for machine collaborators}. \textbf{ADRs} (preferably MADR) serve agents twice: as \emph{input context} (why is the system structured this way? which options were rejected, and why?) and as \emph{output format} -- an agent drafts, a human decides and signs, per Assumption A1 \textcolor{codegray}{\scriptsize (deck 3)}
\item \textbf{Agent instruction files} -- project-local \texttt{CLAUDE.md} and the vendor-neutral \texttt{AGENTS.md} (published 2025, adopted within months by over 60{,}000 open-source repositories) -- carry stack, conventions, build and test commands, module boundaries, no-go zones; loaded at every session start: documentation once ``too expensive to maintain for human readers'' now amortises because it is \emph{executed} on every agent run. \textbf{Machine-checkable conventions} (dependency directions, naming, layering) are a failing test rather than a prose exhortation -- the fitness-function discipline of Part IV
\item \textbf{The corollary cuts both ways}: documentation debt is now reproduced at machine speed -- a stale convention file or ADR is executed by every agent session; DORA 2025: ``AI-accessible internal knowledge'' and healthy data ecosystems rank among the seven capabilities that amplify AI benefits
\end{itemize}
\end{frame}
\begin{frame}{Architecture documentation as a control interface for agents (2/2)}
\footnotesize Excerpt from the course project's agent instruction file (\texttt{AGENTS.md}):
\vspace{0.05cm}
\begin{tcolorbox}[colback=gray!4!white, colframe=gray!55!black, boxrule=0.6pt, arc=2pt, top=3pt, bottom=3pt, left=6pt, right=6pt]
\scriptsize\ttfamily
\# Portfolio Intelligence Platform -- agent instructions\\
\textbf{\#\# Architecture (binding; see docs/adr/)}\\
- Modular monolith, module boundaries enforced by CI\\
~~(see fitness\_functions/boundaries\_test.py). Do not add\\
~~cross-module imports; use the module's public API.\\
- All LLM access goes through gateway/ -- never call a\\
~~provider SDK from domain code (ADR-011).\\
\textbf{\#\# Verification (run before proposing changes)}\\
- make test~~~~~~~~~~\# unit + module-boundary rules\\
- make evals~~~~~~~~~\# eval harness; required for any\\
~~~~~~~~~~~~~~~~~~~~~\# change under prompts/ or gateway/\\
\textbf{\#\# No-go zones}\\
- ledger/ : append-only audit journal. Propose changes\\
~~as an ADR draft instead of editing code.
\end{tcolorbox}
\vspace{0.05cm}
\scriptsize \textcolor{codegray}{Every line is a control statement that an agent executes on each run -- and that therefore must be kept as current as code.}
\end{frame}
\begin{frame}{Guardrails as the precondition for safe agent use}
\footnotesize
\begin{itemize}\setlength\itemsep{4pt}
\item If verification is the scarce resource, then everything that \emph{automates} verification multiplies the value of AI tooling -- and everything that leaves verification informal converts AI speed into instability
\item \textbf{Test suites are the operating licence}: against a dense, fast test suite an agent can iterate -- wrong code fails immediately and is repaired or discarded at machine speed; without that net every agent-generated change ships \emph{unpriced risk} (DORA's ``strong version control and test automation'' amplifier pair)
\item \textbf{Architectural fitness functions fence the structure}: an objective integrity assessment of an architectural characteristic is the machine-readable form of an architecture decision -- dependency rules, cycle checks, module-boundary verification as CI gates were good practice before AI; with agents in the loop they are the mechanism by which an architect constrains \emph{a collaborator who never attends design meetings}
\item \textbf{The delivery pipeline becomes a defence instrument}: static analysis, SAST, dependency and secret scanning, contract tests, progressive delivery move from hygiene to necessity -- the only controls that \emph{scale with generation volume}
\item \textbf{Continuity with Part IV}: nothing here is new machinery -- the measurement contract already demanded executable invariants; Axis A merely adds a new class of change producer whose volume makes the contract \emph{non-optional}
\end{itemize}
\end{frame}
\begin{frame}{The tool landscape, soberly -- and the AI Lens on MCP}
\footnotesize
\begin{itemize}\setlength\itemsep{1pt}
\item Record the landscape \emph{as a geologist records a riverbed} -- evidence of forces, not a map that stays accurate. Generation 2021--2023 (autocomplete-style assistants) suggested lines; generation 2024/2025 onwards \textbf{plans, edits multiple files, runs builds and tests, iterates on failures} -- agentic loops with tool access
\item \textbf{Claude Code} (Anthropic): agentic CLI tool, research preview February 2025, GA May 2025; repository-level anchor \texttt{CLAUDE.md}
\item \textbf{Cursor} (Anysphere): AI-first IDE with an agent mode; the dominant tool among the METR study's experts
\item \textbf{GitHub Copilot}: Copilot Workspace retired May 2025; its concepts live on in the asynchronous \emph{Copilot coding agent} (issues to pull requests, in CI) and the synchronous IDE agent mode
\item \textbf{Devin} (Cognition): ``first AI software engineer'' (2024); 13.86\,\% SWE-bench in March 2024 triggered the agent wave; acquired Windsurf July 2025 -- rapid market consolidation
\item \textbf{Two open standards matter more than any product, because they are architectural}: the \textbf{Model Context Protocol} (MCP, Anthropic, November 2024; JSON-RPC, servers expose tools, resources, prompts; over 10{,}000 public servers) and \textbf{\texttt{AGENTS.md}} for project-level instructions; vendor SDKs extract the agent loop as a library -- the bridge to Axis B (\S44, next week)
\end{itemize}
\vspace{0.05cm}
\begin{ailinse}[MCP is ports-and-adapters at ecosystem scale]
\footnotesize Strip the branding and MCP is a familiar shape: a technology-neutral \emph{port} (the protocol) with swappable \emph{adapters} (servers wrapping databases, ticket systems, browsers), letting any conforming client use any conforming tool -- the role JDBC/ODBC played for databases. \textbf{The hexagonal pattern of Part II did not become obsolete in the agent era; it became an ecosystem standard.}
\end{ailinse}
\end{frame}
\begin{frame}{Benchmarks and their limits -- an expiry date on this section}
\footnotesize
\begin{itemize}\setlength\itemsep{3pt}
\item \textbf{SWE-bench}: 2{,}294 real GitHub issues from twelve Python projects -- given repository and issue text, produce a patch that passes hidden tests. Trajectory: \textbf{1.96\,\%} (best 2023 setup) $\to$ \textbf{13.86\,\%} (Devin, March 2024) $\to$ around \textbf{77--81\,\%} for frontier models by late 2025 on the human-validated 500-task \emph{SWE-bench Verified} subset
\item \textbf{Four qualifications keep the number honest}: (1) \emph{contamination} -- the repositories are in the training data; (2) \emph{scope} -- Python only, issues with tests only; (3) \emph{criterion} -- ``tests pass'' is not ``maintainable, architecture-conformant''; (4) \emph{saturation} -- on the contamination-resistant SWE-bench Pro, frontier models initially scored around 23\,\%. Near-80\,\% benchmark scores next to METR's measured slow-down: \textbf{the module's canonical exercise in benchmark literacy}
\end{itemize}
\vspace{0.1cm}
\begin{hinweisbox}
\footnotesize This section encodes the state of \textbf{early 2026}; product names carry an expiry date measured in months (Copilot Workspace lived roughly a year). Stable -- and \textbf{examinable} -- are the \emph{patterns}: the synchronous pair-agent versus the asynchronous task-agent as interaction modes, context files and ADRs as the control interface, fitness functions as the containment mechanism. Every concrete tool claim carries its own temporal fitness function: \emph{re-verify on every tool generation}.
\end{hinweisbox}
\end{frame}
\begin{frame}{Risks and responsibility: security, bias, skill, accountability}
\footnotesize
\begin{itemize}\setlength\itemsep{2pt}
\item \textbf{Security} -- the evidence predates the agent wave and gains relevance with volume: roughly \textbf{40\,\%} of 1{,}689 Copilot-generated programs (89 security-relevant scenarios) contained CWE top-25 vulnerabilities; a user study: participants with an AI assistant wrote \textbf{less secure code while believing it more secure}; package hallucination (``slopsquatting''): across roughly 576{,}000 generations, \textbf{about a fifth} of recommended package references did not exist -- names an attacker can register pre-emptively. Consequence: SAST, dependency and secret scanning, licence checks in CI are not optional; security review capacity must scale with generation volume
\item \textbf{Automation bias}: over-trust in automated systems is a decades-old human-factors finding -- Perry et al.'s participants overestimated their security, METR's experts overestimated their speed
\item \textbf{Skill formation}: a randomised study of engineers learning a new library -- AI assistance reduced comprehension-test scores by roughly \textbf{17\,\%}; the usage pattern is the decisive moderator (conceptual questions preserved learning, wholesale delegation destroyed it); \textbf{entry-level developer positions are measurably declining}. For this module: the role being trained is the \emph{specifier, verifier, and architect} -- rebuild the competence ladder deliberately, including AI-free practice of fundamentals
\item \textbf{Accountability}: legally and professionally, \emph{the person who merges code answers for it}, regardless of what generated it; AI tools are not liability-bearing entities -- treat AI output as the contribution of an unknown third party: mandatory review, provenance labelling, an explicit policy for permitted uses (DORA 2025: a clearly communicated AI policy first among the seven amplifier capabilities)
\item AI may \emph{draft} an ADR; a nameable person decides, signs, and defends it \textcolor{codegray}{(deck 3)} -- \textcolor{bankblue}{\textbf{architecture is an accountability performance, not a text-production performance}}. IP risk open but manageable: \emph{Doe v.\ GitHub} -- the DMCA claim dismissed in 2024, licence-related claims continue; response: provider duplication filters and indemnification, licence scanning in CI, a documented residual risk in the governance record
\end{itemize}
\end{frame}
\begin{frame}{Project link: Axis A governs how you build the platform}
\begin{projektbox}
\footnotesize Axis A governs \emph{how} you build the Portfolio Intelligence Platform; the project applies every mechanism of this section:
\begin{enumerate}\setlength\itemsep{2pt}
\item[(i)] the repository carries an \texttt{AGENTS.md}\,/\,\texttt{CLAUDE.md} in the spirit of the listing -- and you are expected to \textbf{keep it as current as code}
\item[(ii)] every architecture decision is an \textbf{ADR} -- agents may draft, but a named team member signs
\item[(iii)] agent-generated changes enter the main branch \textbf{only through the CI gate}: module-boundary fitness functions, the test suite, and -- for anything touching prompts or the gateway -- the \textbf{eval harness} of \S42.5
\item[(iv)] your project handbook contains a \textbf{one-page AI policy}: permitted tools, provenance labelling, review rules
\end{enumerate}
\vspace{0.05cm}
\textbf{The graded artefact is not the generated code -- it is the control system around it.}
\end{projektbox}
\end{frame}
\end{document}