% !TEX encoding = UTF-8 Unicode % ============================================================================ % AISE502 -- AI in Software Engineering II % Lecture 12 slides, typeset with the official FHGR beamer theme % (beamerthemeFHGR.sty, University of Applied Sciences of the Grisons). % Slide content is unchanged; only the presentation layer is the FHGR template. % ============================================================================ \documentclass[aspectratio=169]{beamer} \usetheme[showsection, titlebg=pics/theme_pics/titlepage.png]{FHGR} % ============================================ % PACKAGES % The theme already loads tikz, graphicx, xcolor, tabularx, colortbl, % listings, hyperref, environ and xparse -- only the extras are needed here. % ============================================ \usepackage[british]{babel} \usepackage{booktabs} \usepackage{amsmath} \usepackage{amssymb} \usepackage{tcolorbox} \usetikzlibrary{shapes.geometric, arrows.meta, positioning, fit, backgrounds, calc} % ============================================ % SEMANTIC COLOURS, MAPPED ONTO THE FHGR PALETTE % The names used throughout the slides are kept, so no slide text changes; % they now resolve to the FHGR brand colours defined by the theme. % ============================================ \colorlet{bankblue}{blue} % FHGR blue (4B92A4) \colorlet{bankgreen}{green} % FHGR green (817E65) \colorlet{bankred}{red} % FHGR red (C60219) \colorlet{codegray}{gray} % FHGR gray (595959) \colorlet{backcolour}{linen} % FHGR linen (E1D3B5) \definecolor{aiviolet}{HTML}{6B4E71} % muted plum, kept distinct for the AI lens % Attribution labels in English (theme default is German) \renewcommand{\source}[1]{\par\hfill {\tiny\color{FHGRDeco} Source:\,\itshape #1}} \renewcommand{\imagesource}[1]{\par\hfill {\tiny\color{FHGRDeco} Image source:\,\itshape #1}} % ============================================ % CUSTOM TCOLORBOXES (same semantics as the script, FHGR colours) % ============================================ \newtcolorbox{keypoint}{ colback=bankblue!7!white, colframe=bankblue, title=Key Concept, fonttitle=\bfseries\small, boxrule=0.8pt, arc=2pt, top=2pt, bottom=2pt, left=4pt, right=4pt } \newtcolorbox{examplebox}[1][]{ colback=bankgreen!10!white, colframe=bankgreen, title={Example: #1}, fonttitle=\bfseries\small, boxrule=0.8pt, arc=2pt, top=2pt, bottom=2pt, left=4pt, right=4pt } \newtcolorbox{definitionbox}[1][]{ colback=linen!40!white, colframe=camel!85!black, title={Definition: #1}, fonttitle=\bfseries\small, boxrule=0.8pt, arc=2pt, top=2pt, bottom=2pt, left=4pt, right=4pt } \newtcolorbox{thinkbox}{ colback=lightGray!35!white, colframe=darkGray, title=Discussion, fonttitle=\bfseries\small, boxrule=0.8pt, arc=2pt, top=2pt, bottom=2pt, left=4pt, right=4pt } \newtcolorbox{hinweisbox}{ colback=bankred!5!white, colframe=bankred, title=Important Note, fonttitle=\bfseries\small, boxrule=0.8pt, arc=2pt, top=2pt, bottom=2pt, left=4pt, right=4pt } \newtcolorbox{ailinse}[1][]{ colback=aiviolet!7!white, colframe=aiviolet, title={AI Lens: #1}, fonttitle=\bfseries\small, boxrule=0.8pt, arc=2pt, top=2pt, bottom=2pt, left=4pt, right=4pt } \newtcolorbox{projektbox}{ colback=bankblue!4!white, colframe=bankblue!70!black, title=Project Link: Portfolio Intelligence Platform, fonttitle=\bfseries\small, boxrule=0.8pt, arc=2pt, top=2pt, bottom=2pt, left=4pt, right=4pt } % ============================================ % TITLE METADATA % ============================================ \title[AI in Software Engineering II]{AISE502: AI in Software Engineering II} \subtitle{Lecture 12: The AI Dimension I -- Axis A Evidence, Axis B Foundations\\[0.4ex]{\small Script: Part V, Sections 40--41, 42.1--42.5}} \author{Dr.\ Florian Herzog} \shortname{AISE502} \fullname{Fachhochschule Graub\"unden, Chur -- Autumn Semester 2026} \begin{document} % ============================================ % TITLE SLIDE % ============================================ \FHGRTitlePage % ============================================ % AGENDA % ============================================ \begin{frame}{Agenda} \small \begin{enumerate}\setlength\itemsep{1pt} \item \textbf{Two axes, one method} -- Assumption A6 falls due \item \textbf{Axis A}: two contradictory RCTs and the empirical record \item Reconciling the divergence; the verification bottleneck (\textbf{Maxim 7}); Axis A compact \item \textbf{Axis B}: the news-sentiment call, wired the obvious way \item The three component types; why containment: the SE4AI classics \item The reference architecture: \textbf{LLM gateway, queue, ontology guard} \item The eval harness as an engineering artefact \item This week's exercise: \textbf{AdvisorAgent $+$ sub-agents behind the gateway} \end{enumerate} \end{frame} % ============================================ % RECAP % ============================================ \section{Recap} \begin{frame}{Recap: where we are} \footnotesize \begin{itemize}\setlength\itemsep{3pt} \item \textbf{Parts I--IV complete; Lecture 11 closed Part IV}: fitness functions -- three families (dependency gates, budgets, chaos experiments); the four-layer cascade and the eight-row C10 reference contract; cost of change flat \emph{within} / steep \emph{across} architecture boundaries; Conway -- the fit is three-way; six limits; \textbf{Maxim 9} -- Part IV closed \item \textbf{AI Lens threads so far} -- deck 1: the two AI axes and Assumption A6; deck 3: an LLM component stresses D3, D10, D12; an agent drafts the ADR, a human owns the decision; \textbf{ADR-011}: all LLM calls through one gateway port \item \textbf{Deck 6}: C10 profile -- D12 $=$ H, evals as the operative meaning of testability, cost per request; outlook: agent orchestration reuses the catalogue's topologies, workflows before agents ($15\times$ token finding); \textbf{deck 11}: fitness functions are the operating licence for agents (A); eval pass rate and token budget are fitness functions with old mechanics (B) \item \textbf{Today}: Part V redeems A6 systematically -- the two axes as one method; the Axis A evidence and its resolution; the Axis B foundations up to the eval harness \item \textbf{Project}: M4 delivered (deterministic core fully tested and resilient); \textbf{M5 begins} -- AdvisorAgent $+$ 2--3 sub-agents behind the gateway, ontology guard active; the reference architecture today is just-in-time \end{itemize} \end{frame} % ============================================ % TWO AXES, ONE METHOD % ============================================ \section{Two Axes, One Method} \begin{frame}{Part V opens: the promissory note falls due} \emph{\textcolor{bankblue}{Four parts built a complete decision theory without ever making artificial intelligence its subject -- does the construction survive the technology that defines its decade?}} \vspace{0.2cm} \small \begin{itemize}\setlength\itemsep{4pt} \item Not a rhetorical flourish but a \textbf{promissory note falling due}: Part I issued it as \textbf{Assumption A6} -- \emph{AI components extend the quality attribute space but do not change the method}. The bet in two sentences: everything AI does to software engineering can be absorbed by the apparatus you now own \item If AI-bearing systems required a genuinely different method, the bet would be lost -- this part is where the claim must \textbf{survive contact with the evidence} \item Roadmap -- \textbf{cases first, generalisation after}: two contradictory randomised experiments open Axis A (\S41); one concrete LLM call, wired wrongly and then rightly, opens Axis B (\S42); the matrix reading (\S43) and the emergent pattern (\S44) follow next week \end{itemize} \end{frame} \begin{frame}{The two axes of the AI dimension} \begin{columns}[T] \begin{column}{0.55\textwidth} \begin{definitionbox}[The two axes of the AI dimension] \footnotesize \begin{itemize}\setlength\itemsep{2pt} \item \textbf{Axis A -- AI as a tool in the SDLC.} Code assistants, agentic coding tools, review bots participate in \emph{building} the software (code, tests, documentation, design drafts); what ships may contain no AI at all. Unit of analysis: the \emph{development process} and its economics \item \textbf{Axis B -- AI as a runtime component.} LLM services, trained ML models, optimisation solvers are \emph{part of the delivered system} and execute in production. Unit of analysis: the \emph{running system} and its quality attributes \item \textbf{The axes are independent}: a payroll system built with heavy agent support (A without B); a hand-crafted AI-native advisory platform (B without A). In the course project both apply simultaneously -- which is why they must be kept conceptually apart \end{itemize} \end{definitionbox} \end{column} \begin{column}{0.42\textwidth} \begin{center} \resizebox{\linewidth}{!}{% \begin{tikzpicture}[ sysbox/.style={rectangle, draw, rounded corners=4pt, minimum width=3.6cm, minimum height=1.0cm, align=center, font=\small\sffamily, line width=0.8pt}, ax/.style={sysbox, fill=aiviolet!15, draw=aiviolet}, proc/.style={sysbox, fill=bankgreen!15, draw=bankgreen}, core/.style={sysbox, fill=bankblue!20, draw=bankblue, font=\small\sffamily\bfseries}, arr/.style={-{Stealth[length=2.5mm]}, thick, gray!60!black} ] \node[ax] (axisa) at (0,2.4) {\textbf{Axis A}\\AI as tool: agents, assistants}; \node[ax] (axisb) at (5.0,2.4) {\textbf{Axis B}\\AI as component: LLM, ML, solver}; \node[proc] (sdlc) at (0,0) {Development process\\(specify, build, verify, operate)}; \node[core] (system) at (5.0,0) {Delivered system\\(structure, quality attributes)}; \draw[arr] (axisa) -- node[right, font=\scriptsize\sffamily, align=left]{shifts SDLC\\economics} (sdlc); \draw[arr] (axisb) -- node[right, font=\scriptsize\sffamily, align=left]{stretches quality\\attribute space} (system); \draw[arr] (sdlc) -- node[above, font=\scriptsize\sffamily]{produces} (system); \end{tikzpicture}% } \end{center} \vspace{-0.1cm} \scriptsize Axis A changes \emph{how} systems are built; Axis B \emph{what} the built system contains -- both absorbed by the same method: scenarios, tactics, profiles, ADRs, fitness functions. \vspace{0.1cm} \begin{keypoint} \footnotesize \textbf{Two axes, one method}: independent, kept apart -- and both analysed with the apparatus of Parts I--IV, \emph{nothing new}. \end{keypoint} \end{column} \end{columns} \end{frame} % ============================================ % AXIS A -- EVIDENCE AND RESOLUTION % ============================================ \section{Axis A -- Evidence and Resolution} \begin{frame}{Case 1 -- the Copilot RCT: $+55.8\,\%$ on a greenfield task} \emph{\textcolor{bankblue}{AI makes developers 55.8\,\% faster -- or 19\,\% slower. Which study is wrong?}} \vspace{0.1cm} \small Both numbers come from randomised controlled trials, both methodologically sound; few topics in software engineering carry a larger gap between headline and evidence. \vspace{0.15cm} \small \begin{itemize}\setlength\itemsep{2pt} \item \textbf{Case 1 (published 2023)}: 95 professional developers randomly split into two groups; same task -- implement an HTTP server in JavaScript; one group with GitHub Copilot, one without; the clock measured time to completion \item The treatment group finished \textbf{55.8\,\% faster} \item \textbf{Qualification 1}: the confidence interval (\textbf{21--89\,\%}) is very wide -- the headline number is a point estimate, not a natural constant \item \textbf{Qualification 2}: a bounded, well-defined \emph{greenfield} exercise -- no legacy context, no architectural constraints, no review process \item \textbf{Qualification 3}: speed was measured, not quality; completion rates did not differ significantly \item Within those bounds the result is real -- and it is the origin of the ``AI doubles productivity'' headline genre \end{itemize} \end{frame} \begin{frame}{Case 2 -- the METR RCT: $19\,\%$ slower in your own mature codebase} \emph{\textcolor{bankblue}{What happens when the same technology meets experts on their own terrain?}} \vspace{0.15cm} \footnotesize \begin{itemize}\setlength\itemsep{3pt} \item \textbf{16 experienced open-source maintainers}; 246 real issues in repositories they had maintained for years -- large, mature codebases (over a million lines) with high implicit quality standards; each issue randomly assigned to an AI-allowed condition (predominantly Cursor with frontier models of early 2025) or an AI-forbidden condition \item With AI, the developers took \textbf{19\,\% longer} \item \textbf{The perception data are the didactic core}: forecast before the study \textbf{$+24\,\%$} speed-up; measured \textbf{$-19\,\%$}; post-hoc estimate \textbf{$+20\,\%$} faster -- even experts cannot validly introspect their own AI-assisted productivity \item METR's own explanation \emph{maps boundary conditions} rather than refuting Case 1: deep repository familiarity left little for AI-supplied context to add; codebases large and conventionally dense; substantial time spent checking, repairing, discarding AI proposals \end{itemize} \end{frame} \begin{frame}{The full empirical record, 2023--2025} \footnotesize The two cases are the extreme corners of a larger record -- seven strands, 2023--2025, from randomised experiments to organisational telemetry and longitudinal code analysis; read every row \emph{setting first, finding second}. \vspace{-0.1cm} \scriptsize \renewcommand{\arraystretch}{0.8}% \begin{center} \begin{tabular}{@{}p{2.1cm}p{4.3cm}p{7.0cm}@{}} \toprule \textbf{Evidence} & \textbf{Setting} & \textbf{Finding} \\ \midrule Copilot RCT & 95 developers; greenfield HTTP server (JavaScript) & \textbf{$+55.8\,\%$} task speed (95\,\% CI 21--89\,\%); completion rate not significantly different \\ Three field experiments & 4{,}867 developers; Microsoft, Accenture, Fortune-100 firm & \textbf{$+26.1\,\%$} completed tasks (s.e.\ 10.3\,\%); less experienced developers gain most \\ METR RCT & 16 expert OSS maintainers; 246 real issues, own mature repositories & \textbf{19\,\% slower} with AI -- while estimating afterwards that AI had made them 20\,\% faster \\ DORA 2024 & $\sim$3{,}000 respondents; organisational delivery level & $+25\,\%$ AI adoption associated with \textbf{$-1.5\,\%$ throughput} and \textbf{$-7.2\,\%$ delivery stability} \\ DORA 2025 & $\sim$5{,}000 respondents & throughput association now positive; \textbf{instability persists}; AI acts as an \emph{amplifier} of existing strengths and dysfunctions \\ GitClear longitudinal & 211 million changed code lines, 2020--2024 & \textbf{$4\times$} growth in code duplication; moved-code share (the refactoring signature) collapsed from $\sim$25\,\% to below 10\,\% \\ Stack Overflow survey & $>$49{,}000 developers & 84\,\% use or plan to use AI; \textbf{46\,\% actively distrust} its output; top frustration: ``almost right'' code \\ \bottomrule \end{tabular} \end{center} \vspace{-0.15cm} \scriptsize \textcolor{codegray}{Caution from GitHub's own telemetry-plus-survey study: the best predictor of \emph{perceived} productivity is the suggestion acceptance rate, not the persistence of accepted code in the repository -- much vendor-reported ``productivity'' evidence measures perception, not verified output. Case 2's perception gap is the controlled-trial demonstration of the same fact.} \end{frame} \begin{frame}{The system level: DORA 2024 and 2025} \footnotesize \begin{itemize}\setlength\itemsep{1pt} \item DORA measures neither task times nor perceptions but \textbf{delivery performance at the level of the organisation} -- throughput and stability -- exactly the level at which architecture acts \item \textbf{2024} ($\sim$3{,}000 respondents; 75.9\,\% use AI for at least part of their work, roughly three quarters report productivity gains) -- a 25\,\% increase in AI adoption is associated with: \begin{center} \scriptsize \renewcommand{\arraystretch}{0.85}% \begin{tabular}{@{}ll@{}} \toprule \textbf{Gains} & \textbf{Losses} \\ \midrule $+7.5\,\%$ documentation quality & $-1.5\,\%$ delivery throughput \\ $+3.4\,\%$ code quality & $-7.2\,\%$ delivery stability \\ $+3.1\,\%$ review speed & \\ \bottomrule \end{tabular} \end{center} \vspace{-0.1cm} Proposed mechanism is classical: more code per change, and larger batch sizes have been a documented risk driver for years \item \textbf{2025} ($\sim$5{,}000 respondents): adoption near saturation (90\,\%, median about two hours of daily use); more than 80\,\% report productivity gains; 30\,\% still express little or no trust in AI-generated code; the throughput association has \textbf{turned positive} as tools and practices matured -- the negative association with delivery stability \textbf{persists} \item Central metaphor: \textbf{AI is an amplifier} -- it magnifies the strengths of well-run organisations and the dysfunctions of badly run ones \item \emph{Individual acceleration and system-level performance are different quantities, and only the second one pays salaries} \end{itemize} \end{frame} \begin{frame}{Code structure and practitioner trust in the longitudinal record} \footnotesize \begin{itemize}\setlength\itemsep{4pt} \item \textbf{GitClear}, 211 million changed lines (2020--2024): duplicated code blocks (five or more lines) at \textbf{four times} their pre-AI level in 2024; \emph{moved} lines -- the fingerprint of refactoring and modularisation -- fell from roughly \textbf{25\,\% to under 10\,\%}: 2024 was the first year in which copy-paste exceeded code movement \item \textbf{Churn} -- code reworked or discarded within two weeks of commit -- rose from a pre-AI baseline of roughly 3--4\,\% to \textbf{5.7\,\%} in 2024, trend continuing; two obligatory caveats: GitClear is a commercial analytics vendor, and the analysis is \emph{correlational} -- AI's causal share is plausible but not isolated \item Converges with DORA's stability data: \textbf{more code, produced faster, structurally worse maintained} -- reuse by abstraction displaced by reuse by duplication, the opposite of what Parnas-style modularisation (Part II) works to achieve \item \textbf{Stack Overflow 2025} ($>$49{,}000 developers): \textbf{84\,\%} use or plan to use AI tools; \textbf{46\,\%} actively distrust the accuracy of the output; most-cited frustration (45\,\%): ``almost right, but not quite'' -- \emph{adoption rises while trust falls}, consistent with METR and DORA: the effort has migrated from writing to verifying \end{itemize} \end{frame} \begin{frame}{Reconciling the divergence: five moderator variables} \footnotesize \textbf{So which study is wrong? Neither} -- resolving the contradiction \emph{is} the lesson: different populations (task novices vs.\ domain experts in their own code), different codebases (greenfield vs.\ mature), different tasks (bounded vs.\ real issues) -- the results never actually compete; the resolution requires reading \emph{study designs}, not abstracts. Indexed by their moderator variables, Case 1 and Case 2 sit at opposite corners of a five-dimensional design space -- and every other row of the record finds its place in the same coordinates. \vspace{0.1cm} \footnotesize \renewcommand{\arraystretch}{0.9}% \begin{center} \begin{tabular}{@{}p{2.2cm}p{5.0cm}p{5.2cm}@{}} \toprule \textbf{Moderator} & \textbf{Gains high} & \textbf{Gains low or negative} \\ \midrule Experience & juniors, task novices (Copilot RCT, field experiments) & domain experts in their own code (METR) \\ Codebase & greenfield, small, standard stack & mature, large, dense implicit conventions \\ Task & well-defined, bounded & under-specified, cross-cutting \\ Measurement & task time, perceived productivity & delivery stability, maintainability, churn (DORA 2024, GitClear) \\ Organisation & small batches, test automation, loose coupling & large batches, weak guardrails, tight coupling (DORA 2025) \\ \bottomrule \end{tabular} \end{center} \vspace{0.05cm} \footnotesize The same technology yields $+55.8\,\%$ and $-19\,\%$ because the two cases differ on \textbf{every one of the five rows}. \end{frame} \begin{frame}{Discussion: which setting is yours?} \begin{thinkbox} \footnotesize \begin{itemize}\setlength\itemsep{4pt} \item The same class of technology produced $+55.8\,\%$ in one randomised experiment and $-19\,\%$ in another. Walk through the five moderators: \textbf{on which rows do the two studies differ?} \item Consider the systems you are likely to work on two years after graduation -- greenfield exercises, or mature codebases with implicit conventions? \textbf{Which study's setting is closer to that reality?} \item What does the METR perception gap (forecast $+24\,\%$, measured $-19\,\%$, post-hoc estimate $+20\,\%$) imply about \textbf{relying on your own felt productivity as evidence}? \end{itemize} \end{thinkbox} \vspace{0.2cm} \scriptsize \textcolor{codegray}{\textbf{Project transfer:} which moderator row describes your repository in week 12?} \end{frame} \begin{frame}{The verification bottleneck} \small The structural conclusion underneath the moderator table, in one sentence: \vspace{0.1cm} \begin{center} \textbf{Code generation became cheap; specification, verification, and architecture became the binding constraints.} \end{center} \vspace{0.1cm} \small \begin{itemize}\setlength\itemsep{4pt} \item When the marginal cost of producing plausible code approaches zero, the scarce resource is no longer typing but \textbf{everything that surrounds it}: understanding the requirement precisely enough to specify it, reviewing and testing what was generated, and accepting responsibility for shipping it \item \textbf{The strands converge}: DORA -- individual acceleration coexists with delivery instability where control systems are weak; two thirds of surveyed developers spend more time on almost-right code; a substantial share of METR's slow-down is time spent checking, repairing, discarding AI proposals; industry analyses describe \emph{code review as the new bottleneck} -- more and larger pull requests meeting unchanged human review capacity \item Economically put: \textbf{AI lowers the cost of \emph{producing} code, not the cost of \emph{taking responsibility} for code} \end{itemize} \end{frame} \begin{frame}{Three consequences bind Axis A into the fit theory -- Maxim 7} \footnotesize \begin{enumerate}\setlength\itemsep{2pt} \item \textbf{Architecture quality gates AI gains} -- DORA 2025's core finding: teams in loosely coupled architectures with fast feedback loops convert AI adoption into throughput; tightly coupled systems with slow processes do not -- the AI-era echo of loosely coupled architectures and teams as the strongest predictor of continuous delivery performance \textcolor{codegray}{(the coupling finding of Lecture 11)}. In the theory's vocabulary: \textbf{D7} (evolvability) and \textbf{D9} (testability and deployability) gain weight in \emph{every} requirements profile -- architecture--application fit acquires a second reading: fit to a \emph{mode of work} in which change volume rises by an order of magnitude \item \textbf{Architecture documentation becomes a control interface} -- ADRs, repository convention files, and machine-readable rules are no longer passive records; agents execute them on every run (\S41.6) \item \textbf{Fitness functions become the operating licence for agents} -- an agent iterating against a dense test suite and CI-enforced architecture rules is contained; without them, every agent change is unpriced risk (\S41.7) \end{enumerate} \vspace{0.1cm} \begin{keypoint} \footnotesize \textbf{Maxim 7.} Good architecture was always the art of making change cheap and safe; AI raises the change rate by an order of magnitude -- and therefore \textbf{raises, not lowers, the value of architecture}. \end{keypoint} \end{frame} % ============================================ % AXIS A -- CONTROL INTERFACE AND GUARDRAILS % ============================================ \section{Axis A -- Control Interface and Guardrails} \begin{frame}{Architecture documentation as a control interface for agents (1/2)} \footnotesize \begin{itemize}\setlength\itemsep{5pt} \item Agentic tools are \textbf{context-driven}: they produce architecture-conformant code only if the architecture is \emph{explicit, machine-readable, and in the repository} -- Part I's documentation artefacts, written for human readers, upgrade into a \textbf{control interface for machine collaborators}. \textbf{ADRs} (preferably MADR) serve agents twice: as \emph{input context} (why is the system structured this way? which options were rejected, and why?) and as \emph{output format} -- an agent drafts, a human decides and signs, per Assumption A1 \textcolor{codegray}{\scriptsize (deck 3)} \item \textbf{Agent instruction files} -- project-local \texttt{CLAUDE.md} and the vendor-neutral \texttt{AGENTS.md} (published 2025, adopted within months by over 60{,}000 open-source repositories) -- carry stack, conventions, build and test commands, module boundaries, no-go zones; loaded at every session start: documentation once ``too expensive to maintain for human readers'' now amortises because it is \emph{executed} on every agent run. \textbf{Machine-checkable conventions} (dependency directions, naming, layering) are a failing test rather than a prose exhortation -- the fitness-function discipline of Part IV \item \textbf{The corollary cuts both ways}: documentation debt is now reproduced at machine speed -- a stale convention file or ADR is executed by every agent session; DORA 2025: ``AI-accessible internal knowledge'' and healthy data ecosystems rank among the seven capabilities that amplify AI benefits \end{itemize} \end{frame} \begin{frame}{Architecture documentation as a control interface for agents (2/2)} \footnotesize Excerpt from the course project's agent instruction file (\texttt{AGENTS.md}): \vspace{0.05cm} \begin{tcolorbox}[colback=gray!4!white, colframe=gray!55!black, boxrule=0.6pt, arc=2pt, top=3pt, bottom=3pt, left=6pt, right=6pt] \scriptsize\ttfamily \# Portfolio Intelligence Platform -- agent instructions\\ \textbf{\#\# Architecture (binding; see docs/adr/)}\\ - Modular monolith, module boundaries enforced by CI\\ ~~(see fitness\_functions/boundaries\_test.py). Do not add\\ ~~cross-module imports; use the module's public API.\\ - All LLM access goes through gateway/ -- never call a\\ ~~provider SDK from domain code (ADR-011).\\ \textbf{\#\# Verification (run before proposing changes)}\\ - make test~~~~~~~~~~\# unit + module-boundary rules\\ - make evals~~~~~~~~~\# eval harness; required for any\\ ~~~~~~~~~~~~~~~~~~~~~\# change under prompts/ or gateway/\\ \textbf{\#\# No-go zones}\\ - ledger/ : append-only audit journal. Propose changes\\ ~~as an ADR draft instead of editing code. \end{tcolorbox} \vspace{0.05cm} \scriptsize \textcolor{codegray}{Every line is a control statement that an agent executes on each run -- and that therefore must be kept as current as code.} \end{frame} \begin{frame}{Guardrails as the precondition for safe agent use} \footnotesize \begin{itemize}\setlength\itemsep{4pt} \item If verification is the scarce resource, then everything that \emph{automates} verification multiplies the value of AI tooling -- and everything that leaves verification informal converts AI speed into instability \item \textbf{Test suites are the operating licence}: against a dense, fast test suite an agent can iterate -- wrong code fails immediately and is repaired or discarded at machine speed; without that net every agent-generated change ships \emph{unpriced risk} (DORA's ``strong version control and test automation'' amplifier pair) \item \textbf{Architectural fitness functions fence the structure}: an objective integrity assessment of an architectural characteristic is the machine-readable form of an architecture decision -- dependency rules, cycle checks, module-boundary verification as CI gates were good practice before AI; with agents in the loop they are the mechanism by which an architect constrains \emph{a collaborator who never attends design meetings} \item \textbf{The delivery pipeline becomes a defence instrument}: static analysis, SAST, dependency and secret scanning, contract tests, progressive delivery move from hygiene to necessity -- the only controls that \emph{scale with generation volume} \item \textbf{Continuity with Part IV}: nothing here is new machinery -- the measurement contract already demanded executable invariants; Axis A merely adds a new class of change producer whose volume makes the contract \emph{non-optional} \end{itemize} \end{frame} \begin{frame}{The tool landscape, soberly -- and the AI Lens on MCP} \footnotesize \begin{itemize}\setlength\itemsep{1pt} \item Record the landscape \emph{as a geologist records a riverbed} -- evidence of forces, not a map that stays accurate. Generation 2021--2023 (autocomplete-style assistants) suggested lines; generation 2024/2025 onwards \textbf{plans, edits multiple files, runs builds and tests, iterates on failures} -- agentic loops with tool access \item \textbf{Claude Code} (Anthropic): agentic CLI tool, research preview February 2025, GA May 2025; repository-level anchor \texttt{CLAUDE.md} \item \textbf{Cursor} (Anysphere): AI-first IDE with an agent mode; the dominant tool among the METR study's experts \item \textbf{GitHub Copilot}: Copilot Workspace retired May 2025; its concepts live on in the asynchronous \emph{Copilot coding agent} (issues to pull requests, in CI) and the synchronous IDE agent mode \item \textbf{Devin} (Cognition): ``first AI software engineer'' (2024); 13.86\,\% SWE-bench in March 2024 triggered the agent wave; acquired Windsurf July 2025 -- rapid market consolidation \item \textbf{Two open standards matter more than any product, because they are architectural}: the \textbf{Model Context Protocol} (MCP, Anthropic, November 2024; JSON-RPC, servers expose tools, resources, prompts; over 10{,}000 public servers) and \textbf{\texttt{AGENTS.md}} for project-level instructions; vendor SDKs extract the agent loop as a library -- the bridge to Axis B (\S44, next week) \end{itemize} \vspace{0.05cm} \begin{ailinse}[MCP is ports-and-adapters at ecosystem scale] \footnotesize Strip the branding and MCP is a familiar shape: a technology-neutral \emph{port} (the protocol) with swappable \emph{adapters} (servers wrapping databases, ticket systems, browsers), letting any conforming client use any conforming tool -- the role JDBC/ODBC played for databases. \textbf{The hexagonal pattern of Part II did not become obsolete in the agent era; it became an ecosystem standard.} \end{ailinse} \end{frame} \begin{frame}{Benchmarks and their limits -- an expiry date on this section} \footnotesize \begin{itemize}\setlength\itemsep{3pt} \item \textbf{SWE-bench}: 2{,}294 real GitHub issues from twelve Python projects -- given repository and issue text, produce a patch that passes hidden tests. Trajectory: \textbf{1.96\,\%} (best 2023 setup) $\to$ \textbf{13.86\,\%} (Devin, March 2024) $\to$ around \textbf{77--81\,\%} for frontier models by late 2025 on the human-validated 500-task \emph{SWE-bench Verified} subset \item \textbf{Four qualifications keep the number honest}: (1) \emph{contamination} -- the repositories are in the training data; (2) \emph{scope} -- Python only, issues with tests only; (3) \emph{criterion} -- ``tests pass'' is not ``maintainable, architecture-conformant''; (4) \emph{saturation} -- on the contamination-resistant SWE-bench Pro, frontier models initially scored around 23\,\%. Near-80\,\% benchmark scores next to METR's measured slow-down: \textbf{the module's canonical exercise in benchmark literacy} \end{itemize} \vspace{0.1cm} \begin{hinweisbox} \footnotesize This section encodes the state of \textbf{early 2026}; product names carry an expiry date measured in months (Copilot Workspace lived roughly a year). Stable -- and \textbf{examinable} -- are the \emph{patterns}: the synchronous pair-agent versus the asynchronous task-agent as interaction modes, context files and ADRs as the control interface, fitness functions as the containment mechanism. Every concrete tool claim carries its own temporal fitness function: \emph{re-verify on every tool generation}. \end{hinweisbox} \end{frame} \begin{frame}{Risks and responsibility: security, bias, skill, accountability} \footnotesize \begin{itemize}\setlength\itemsep{2pt} \item \textbf{Security} -- the evidence predates the agent wave and gains relevance with volume: roughly \textbf{40\,\%} of 1{,}689 Copilot-generated programs (89 security-relevant scenarios) contained CWE top-25 vulnerabilities; a user study: participants with an AI assistant wrote \textbf{less secure code while believing it more secure}; package hallucination (``slopsquatting''): across roughly 576{,}000 generations, \textbf{about a fifth} of recommended package references did not exist -- names an attacker can register pre-emptively. Consequence: SAST, dependency and secret scanning, licence checks in CI are not optional; security review capacity must scale with generation volume \item \textbf{Automation bias}: over-trust in automated systems is a decades-old human-factors finding -- Perry et al.'s participants overestimated their security, METR's experts overestimated their speed \item \textbf{Skill formation}: a randomised study of engineers learning a new library -- AI assistance reduced comprehension-test scores by roughly \textbf{17\,\%}; the usage pattern is the decisive moderator (conceptual questions preserved learning, wholesale delegation destroyed it); \textbf{entry-level developer positions are measurably declining}. For this module: the role being trained is the \emph{specifier, verifier, and architect} -- rebuild the competence ladder deliberately, including AI-free practice of fundamentals \item \textbf{Accountability}: legally and professionally, \emph{the person who merges code answers for it}, regardless of what generated it; AI tools are not liability-bearing entities -- treat AI output as the contribution of an unknown third party: mandatory review, provenance labelling, an explicit policy for permitted uses (DORA 2025: a clearly communicated AI policy first among the seven amplifier capabilities) \item AI may \emph{draft} an ADR; a nameable person decides, signs, and defends it \textcolor{codegray}{(deck 3)} -- \textcolor{bankblue}{\textbf{architecture is an accountability performance, not a text-production performance}}. IP risk open but manageable: \emph{Doe v.\ GitHub} -- the DMCA claim dismissed in 2024, licence-related claims continue; response: provider duplication filters and indemnification, licence scanning in CI, a documented residual risk in the governance record \end{itemize} \end{frame} \begin{frame}{Project link: Axis A governs how you build the platform} \begin{projektbox} \footnotesize Axis A governs \emph{how} you build the Portfolio Intelligence Platform; the project applies every mechanism of this section: \begin{enumerate}\setlength\itemsep{2pt} \item[(i)] the repository carries an \texttt{AGENTS.md}\,/\,\texttt{CLAUDE.md} in the spirit of the listing -- and you are expected to \textbf{keep it as current as code} \item[(ii)] every architecture decision is an \textbf{ADR} -- agents may draft, but a named team member signs \item[(iii)] agent-generated changes enter the main branch \textbf{only through the CI gate}: module-boundary fitness functions, the test suite, and -- for anything touching prompts or the gateway -- the \textbf{eval harness} of \S42.5 \item[(iv)] your project handbook contains a \textbf{one-page AI policy}: permitted tools, provenance labelling, review rules \end{enumerate} \vspace{0.05cm} \textbf{The graded artefact is not the generated code -- it is the control system around it.} \end{projektbox} \end{frame} \end{document}