diff --git a/.obsidian/graph.json b/.obsidian/graph.json index a889a2f..3149205 100644 --- a/.obsidian/graph.json +++ b/.obsidian/graph.json @@ -17,6 +17,6 @@ "repelStrength": 10, "linkStrength": 1, "linkDistance": 250, - "scale": 0.999999999999998, + "scale": 0.5087618855792575, "close": false } \ No newline at end of file diff --git a/.obsidian/workspace.json b/.obsidian/workspace.json new file mode 100644 index 0000000..7d98bd2 --- /dev/null +++ b/.obsidian/workspace.json @@ -0,0 +1,225 @@ +{ + "main": { + "id": "0ddc67276227899d", + "type": "split", + "children": [ + { + "id": "898d980991f1e045", + "type": "tabs", + "children": [ + { + "id": "5463f555de76b7cd", + "type": "leaf", + "state": { + "type": "markdown", + "state": { + "file": "index.md", + "mode": "source", + "source": false + }, + "icon": "lucide-file", + "title": "index" + } + } + ] + } + ], + "direction": "vertical" + }, + "left": { + "id": "bdbe0fd792c2e4fb", + "type": "split", + "children": [ + { + "id": "e9a3ec3578cd4638", + "type": "tabs", + "children": [ + { + "id": "d3f791c8588a6b77", + "type": "leaf", + "state": { + "type": "file-explorer", + "state": { + "sortOrder": "alphabetical", + "autoReveal": false + }, + "icon": "lucide-folder-closed", + "title": "Files" + } + }, + { + "id": "474340f2dfe84c24", + "type": "leaf", + "state": { + "type": "search", + "state": { + "query": "complain", + "matchingCase": false, + "explainSearch": false, + "collapseAll": false, + "extraContext": false, + "sortOrder": "alphabetical" + }, + "icon": "lucide-search", + "title": "Search" + } + }, + { + "id": "874a5b4f1ccbbe51", + "type": "leaf", + "state": { + "type": "bookmarks", + "state": {}, + "icon": "lucide-bookmark", + "title": "Bookmarks" + } + } + ] + } + ], + "direction": "horizontal", + "width": 397.5 + }, + "right": { + "id": "e2dfddd4b270c511", + "type": "split", + "children": [ + { + "id": "7b6dd27eb695f3a5", + "type": "tabs", + "children": [ + { + "id": "56703b36a8b20302", + "type": "leaf", + "state": { + "type": "backlink", + "state": { + "file": "wiki/concepts/solve-first-then-skillify.md", + "collapseAll": false, + "extraContext": false, + "sortOrder": "alphabetical", + "showSearch": false, + "searchQuery": "", + "backlinkCollapsed": false, + "unlinkedCollapsed": true + }, + "icon": "links-coming-in", + "title": "Backlinks for solve-first-then-skillify" + } + }, + { + "id": "61e7b50a5685cf74", + "type": "leaf", + "state": { + "type": "outgoing-link", + "state": { + "file": "wiki/concepts/solve-first-then-skillify.md", + "linksCollapsed": false, + "unlinkedCollapsed": true + }, + "icon": "links-going-out", + "title": "Outgoing links from solve-first-then-skillify" + } + }, + { + "id": "8cc8fddecfb11df6", + "type": "leaf", + "state": { + "type": "tag", + "state": { + "sortOrder": "frequency", + "useHierarchy": true, + "showSearch": false, + "searchQuery": "" + }, + "icon": "lucide-tags", + "title": "Tags" + } + }, + { + "id": "27970b943d6a03d7", + "type": "leaf", + "state": { + "type": "all-properties", + "state": { + "sortOrder": "frequency", + "showSearch": false, + "searchQuery": "" + }, + "icon": "lucide-archive", + "title": "All properties" + } + }, + { + "id": "2c3dedf9326d2711", + "type": "leaf", + "state": { + "type": "outline", + "state": { + "file": "wiki/concepts/solve-first-then-skillify.md", + "followCursor": false, + "showSearch": false, + "searchQuery": "" + }, + "icon": "lucide-list", + "title": "Outline of solve-first-then-skillify" + } + } + ] + } + ], + "direction": "horizontal", + "width": 300, + "collapsed": true + }, + "left-ribbon": { + "hiddenItems": { + "switcher:Open quick switcher": false, + "graph:Open graph view": false, + "canvas:Create new canvas": false, + "daily-notes:Open today's daily note": false, + "templates:Insert template": false, + "command-palette:Open command palette": false, + "bases:Create new base": false + } + }, + "active": "5463f555de76b7cd", + "lastOpenFiles": [ + "wiki/queries/2026-07-28-webinar-theses.md", + "wiki/queries/2026-07-22-webinar-theses.md", + "raw/notes/my theses.md", + "raw/notes/Webinar Plan - From Chat Box to Your Own OS.md", + "raw/notes/Webinar script.md", + "raw/sources/А что если наВайб-Кодить.md", + "index.md.tmp.28724.8b2cd0672e68", + "index.md.tmp.28724.bf94cff6ee0a", + "wiki/overview.md.tmp.28724.772546efd332", + "wiki/overview.md.tmp.28724.8c8f2e4cf507", + "wiki/overview.md.tmp.28724.0d66b3b4bb53", + "wiki/overview.md.tmp.28724.026fed72c5e1", + "wiki/overview.md.tmp.28724.331375f61a61", + "wiki/concepts/code-as-throwaway.md.tmp.28724.a614242c473d", + "wiki/concepts/code-as-throwaway.md.tmp.28724.296cfad627cf", + "wiki/concepts/emacsification-of-software.md.tmp.28724.4ff35ff220f6", + "wiki/concepts/maintenance-is-the-real-cost.md", + "wiki/sources/2026-07-29-what-if-we-vibe-code-it.md", + "raw/sources/In 1 Year, the Gap Between AI Users and Everyone Else Will Be Irreversible.md", + "wiki/queries/2026-07-28-verification-beat-design.md", + "wiki/concepts/enterprise-ai-reality.md", + "wiki/concepts/code-as-throwaway.md", + "wiki/lint-reports/2026-07-28-lint.md", + "wiki/concepts/emacsification-of-software.md", + "raw/sources/Agentic Engineering, explained by a 10x developer.md", + "wiki/concepts/async-by-default.md", + "wiki/concepts/explosion-of-internal-software.md", + "wiki/concepts/build-for-the-agent-not-the-human.md", + "wiki/concepts/shedding-weight.md", + "wiki/entities/amp.md", + "wiki/entities/thorsten-ball.md", + "wiki/sources/2026-07-28-agentic-engineering-10x-developer.md", + "dashboard.md", + "mockups/README.md", + "wiki/concepts/evolution-of-agent-tooling.md", + "wiki/script-coverage.md" + ] +} \ No newline at end of file diff --git a/index.md b/index.md index 4efddbb..9f96ddb 100644 --- a/index.md +++ b/index.md @@ -17,8 +17,10 @@ Every wiki page carries a page-type tag under its H1 (`#source`, `#entity`, `#co - [[2026-07-21-larysa-interview]] — Eugene + Larysa (BA/PM): memory loss, integration dead-ends, less room for imagination _(raw: Larysa interview.md)_ - [[2026-07-22-ai-is-stupid]] — YouTube short (author unknown): "stupid AI" = model minus context minus harness; Nobel-vs-employee analogy _(raw: ИИ глупый!.md)_ - [[2026-07-24-youre-reading-way-too-much-code]] — Theo Browne (video): make more cheap code, four tiers of code, 100:1 slop-to-ship verification _(raw: You're reading way too much code.md)_ +- [[2026-07-28-agentic-engineering-10x-developer]] — Thorsten Ball (AMP): shed weight, orbs/async, Emacsification, internal software; **rejects skills/MCP** _(raw: Agentic Engineering, explained by a 10x developer.md)_ +- [[2026-07-29-what-if-we-vibe-code-it]] — YouTube video (author unknown): maintenance is the real cost, internal service = second business, Jira→Linear pendulum, build-vs-buy checklist _(raw: А что если наВайб-Кодить.md)_ -**Raw, not yet ingested:** `raw/sources/Ideas for webinar.md` · `raw/sources/Webinar Plan - From Chat Box to Your Own OS.md` · `raw/sources/Webinar script.md` · `raw/sources/my theses.md` +**All of `raw/sources/` is ingested.** The webinar-deliverable working notes now live in `raw/notes/` (`Ideas for webinar.md` · `Webinar Plan - From Chat Box to Your Own OS.md` · `Webinar script.md` · `my theses.md` · `introduction.md`) — treated as authored deliverables rather than sources, cited as raw where used. ## Entities @@ -31,9 +33,11 @@ Every wiki page carries a page-type tag under its H1 (`#source`, `#entity`, `#co - [[nina]] — HR recruiter at Virtido; webinar-audience proxy, use-case supplier - [[yulia]] — HR/recruiting lead; webinar organizer (name/affiliation tentative) - [[larysa]] — technical BA/PM, ex-mobile dev; advanced user blocked by memory + integrations +- [[thorsten-ball]] — founding engineer at AMP; 99% AI-written code, no skills/MCP ### Tools / Orgs - [[claude-code]] — reference harness (all sources) +- [[amp]] — Sourcegraph's agent; orbs, Oracle/Painter/Puck, vendor-curated harness - [[hermes]] — skills-first, self-curating harness - [[virtido]] — Sebastian's outsourcing company; its HR team is the webinar audience - [[inspectron]] — Eugene's employer (Edge Compute / IoT) @@ -51,6 +55,7 @@ Every wiki page carries a page-type tag under its H1 (`#source`, `#entity`, `#co - [[solve-first-then-skillify]] — solve the task once, then freeze it into a skill - [[integration-dead-ends]] — agent starts work against connectors the user's account doesn't have - [[leave-less-room-for-imagination]] — every gap in a spec gets filled, invisibly; tighten it +- [[async-by-default]] — orbs/remote sandboxes; one URL = thread + agent + computation + diff; ask for proof **Human side** - [[product-ownership]] — own outcomes, frame problems not tickets @@ -64,6 +69,11 @@ Every wiki page carries a page-type tag under its H1 (`#source`, `#entity`, `#co - [[code-as-throwaway]] — cost of code → zero - [[make-more-cheap-code]] — Theo: four tiers of code; generate never-shipped slop to verify/explore; there's always another layer - [[enterprise-ai-reality]] — compliance lock-down; the company-managed-harness market +- [[shedding-weight]] — delete what only existed because humans were the bottleneck (backlogs, redundant CI, admin panels) +- [[build-for-the-agent-not-the-human]] — no forms, no admin panels, bring your own agent +- [[emacsification-of-software]] — fork and remix rather than upstream; software becomes bespoke +- [[explosion-of-internal-software]] — the Excel/wiki/hack layer becomes real tools; skill + token budget as the divide +- [[maintenance-is-the-real-cost]] — writing was never the bottleneck; internal service = second business; build-vs-buy checklist ## Timelines @@ -77,9 +87,11 @@ Every wiki page carries a page-type tag under its H1 (`#source`, `#entity`, `#co - [[2026-07-14-best-first-skill-for-beginner]] — best first skill for a Claude beginner: skill-creator (meta) + tone-of-voice/anti-AI-language; foundation docs first - [[2026-07-14-network-from-standing-start]] — network-from-zero: tentative protocol + Sebastian round-2 interview instrument (10 questions) -- [[2026-07-22-webinar-theses]] — 14 candidate theses for the webinar, grouped spine / stakes / obstacles / method / tensions +- [[2026-07-22-webinar-theses]] — _superseded_ → 14 candidate theses (7-source state) +- [[2026-07-28-webinar-theses]] — **v2, current**: 17 theses from all 10 sources; T3 reframed (*authored beats inferred*), verification / shedding-weight / ask-for-15-options added; 30-min cut + gaps in the current script +- [[2026-07-28-verification-beat-design]] — how to add the missing verification beat: checker skill, the "right shelf" ambiguity demo, and the closing what-stays-yours half; drafted script copy - [[2026-07-24-non-engineer-throwaway-verification]] — non-engineer analog of throwaway verification code: generated checks not content; checker skills; drift as diagnostic ## Lint Reports -_None yet._ +- [[2026-07-28-lint]] — first pass: structurally clean (0 broken links, 0 orphans, 0 template/tag errors); 9 fixes applied (stale `raw/` paths, 4 one-sided contradictions, 2 staleness notes); open: a `taste` concept page, refreshing the webinar theses, weakly-linked query pages diff --git a/log.md b/log.md index a9be385..b3eac0e 100644 --- a/log.md +++ b/log.md @@ -147,3 +147,86 @@ Entry format: - Pages updated: [[make-more-cheap-code]] (next-question marked answered, link added), [[leave-less-room-for-imagination]] (new next-question: drift harnessed as a sandboxed ambiguity diagnostic — synthesis, tentative), `index.md` (Queries section). - Notes: Pure synthesis, no new source — every mapped pattern is grounded in corpus use cases (Nina/Yulia job descriptions, candidate KB, sourcing; Larysa spec ambiguity; Allie's anti-AI-language checker-skill precedent). Key reframes with durable value: (1) the fresh-agent misread test inverts [[leave-less-room-for-imagination]] — drift becomes a diagnostic when sandboxed; (2) [[solve-first-then-skillify]] populates skills in two species, producers and checkers. Flagged honestly: the whole mapping is argument-by-analogy with no measured claim; Nina's transcript-beats-summary finding is a standing counterweight to summary-based review. - Next: decide whether "checker skills" earns a slide on the webinar's skills rung; the webinar-deliverable docs (`Ideas`, `Plan`, `script`, `my theses.md`) remain un-ingested; first lint pass still overdue (9 sources, 18 concepts, 2026-07-24 now has 1 query, 0 lint reports). + +## 2026-07-28 — ingest (Agentic Engineering, explained by a 10x developer) +- Intent: ingest +- Input: "ingest \"Agentic Engineering, explained by a 10x developer\"" → `raw/sources/Agentic Engineering, explained by a 10x developer.md` (conclusions doc for a 42:33 YouTube interview; Thorsten Ball × David Andre). The densest single source in the vault so far. +- Pages created — source (1): [[2026-07-28-agentic-engineering-10x-developer]]. +- Pages created — entities (2): [[thorsten-ball]] (founding engineer at AMP; author of the Go interpreter/compiler books), [[amp]] (Sourcegraph's agent — the vault's second reference harness). +- Pages created — concepts (5): [[shedding-weight]] (delete process that only existed because humans were the bottleneck — backlogs, CI that repeats the agent's tests, IDE extensions, local dev), [[build-for-the-agent-not-the-human]] (no forms; the admin panel that dies; bring your own agent), [[emacsification-of-software]] (fork-and-remix, never upstream; software becomes bespoke), [[explosion-of-internal-software]] (the Excel/wiki/hack layer becomes real tools; skill + token budget as the two dividing variables), [[async-by-default]] (orbs as remote sandboxes; one URL = thread + agent + computation + diff; ask for proof since you're waiting anyway). +- Pages updated — concepts (10): [[skills-as-memory]] (**the dissent**, plus three competing readings under Contradictions), [[evolution-of-agent-tooling]] ("a fourth position: skip the progression"; A2A partially answered by AMP's agent-to-agent messaging), [[harness]] (AMP as second reference harness; vendor-managed as a third governance option; local-vs-remote tension), [[context-as-scarce-resource]] (information > tuning; the two information sources; context as scarce per-*wallet*), [[code-as-throwaway]] (99%-AI-written production datapoint; slop-is-human), [[make-more-cheap-code]] (variations-not-answers; ask-for-proof; attention-vs-token-budget sharpening), [[product-ownership]] (first-principles as the top skill; the printer/tablet push-back; what got commoditised), [[seniority-and-the-junior-squeeze]] (2–3 years of hand-taught knowledge → 30-second output; "don't compare yourself to the 1%"), [[enterprise-ai-reality]] (token budget as a second divide; why the frontier playbook doesn't transfer), [[leave-less-room-for-imagination]] (his 5-part prompt structure as a worked example; model-choice tension), [[personal-ai-operating-system]] (the fourth layer: tools you build for yourself). +- Pages updated — entities (2): [[theo-browne]] (ally cross-link), [[claude-code]] (AMP as the contrasting harness design). Plus [[overview]] (9→10 sources; new "frontier side" of the through-line; agree/diverge rewritten; 3 new vault-level open questions) and `index.md`. +- Notes: **The important thing this source does is contradict the vault's spine.** Thorsten ships a 99%-AI-written codebase with no skills, no MCP servers and no slash commands — his context lives in the codebase and `AGENTS.md`. This is the first credible rejection of the mechanism the webinar's central promise rests on, and the evidential asymmetry favours him (first-hand daily practice at scale vs Konstantin's architecture argument plus self-reported individual workflows). Recorded as three competing readings on [[skills-as-memory]] — situational (he owns one codebase; the HR/BA audience owns none) / premature abstraction / same thing under another name (AMP's vendor-curated sub-agents and `AGENTS.md` *are* two-stage context, just not user-authored). Not smoothed, not resolved. Second-order tensions logged: model choice as lever (Eugene 4.7-over-4.8) vs distraction (Thorsten); local consolidated workspace (Eugene) vs local dev disappearing (Thorsten); shed-your-process (frontier) vs compliance-is-the-deliverable ([[enterprise-ai-reality]]). +- Also strongly *confirmatory*: the club food-ordering app (menu photo → working app, then ~2 hours of phone typing to encode a 20-person club's ordering process) is the corpus's best outside evidence for the webinar's "little tools you make for yourself" thesis — reached independently, by an engineer, applied to non-technical people. And "slop comes from humans" converges with Theo from a different direction (taste vs verification discipline). +- New named variable with no answer anywhere in the corpus: **token budget** as one of two winner/loser variables. Elevated to a vault-level open question. +- All quantities self-reported from inside the company selling the agent (99% figure, the team poll, velocity claims) — flagged tentative throughout. "Local dev is going away" is a prediction from a remote-sandbox vendor — flagged. +- Housekeeping: corrected the stale `index.md` "not yet ingested" list — those four webinar docs moved to `raw/notes/` (commit 1361dd7) and are authored deliverables, not sources. `raw/sources/` is now fully ingested (10/10). +- Next: the highest-value move is now a **lint pass** — 10 sources, 24 concepts, 0 lint reports, and this ingest added five concepts plus a live contradiction that touches the spine. Candidate query with real stakes for the webinar: *does the skills rung survive Thorsten's counter-example, and what would test it?* (the cheap experiment — same task with and without a skill, in a non-engineer's hands — has never been run). Also worth considering: a comparison page "Thorsten vs the skills camp", and whether [[shedding-weight]] / [[explosion-of-internal-software]] earn stations on the webinar spine. + +## 2026-07-28 — lint (first pass) +- Intent: lint +- Input: "lint wiki" — first health check on the vault (10 sources · 24 concepts · 14 entities · 4 queries · 1 comparison · 1 timeline; 0 prior lint reports). +- Pages created: [[2026-07-28-lint]] (first `wiki/lint-reports/` page). +- Checks run: page-type tags vs folder + line-3 placement (41/41 pass) · H1 presence (41/41) · required template sections for sources (10/10) and entity/concept pages (38/38) · Evidence sections citing ≥1 source page (38/38) · broken `[[links]]` (0) · orphans (0) · zero-outbound pages (0) · `index.md` ↔ `wiki/concepts/` drift (0, 24/24 in sync) · `raw/` path resolution (5 failures) · contradiction reciprocity · high-mention concepts lacking pages · link-graph inbound/outbound counts. +- Headline: **structurally clean; the real defects were all staleness of synthesis.** 12 findings, 9 fixed in this pass, 3 left as recommendations because they are scope decisions rather than defects. +- Fixes applied — references (3): stale `raw/sources/` paths for the four deliverable docs that moved to `raw/notes/` in commit 1361dd7, corrected in [[levels-of-ai-usage]], [[2026-07-22-webinar-theses]] and [[2026-07-14-sebastian-eugene-interview]] (5 refs, reworded "not yet ingested" → "authored deliverable"); [[claude-code]] summary said "cited across all four ingested sources" (written at 4 sources, now 10) — rewritten to a claim that won't rot, noting Thorsten as the sole practitioner on a different harness; [[overview]] open question naming `HR Contacts` (a file that does not exist — corrected in `index.md` on 2026-07-21 but never propagated here) marked resolved. +- Fixes applied — one-sided contradictions (4): rule 5 requires contradictions be recorded explicitly, and four were logged on one page but not on the page holding the opposing view. Added reciprocal entries to [[think-wider-not-bigger]] (vs tight specs), [[levels-of-ai-usage]] (the skills dissent — its *top rung* is what Thorsten contests), [[personal-ai-operating-system]] (memory-as-anti-feature vs its *layer 1*, plus the skills dissent vs its layer 2), [[skills-as-memory]] (skills-vs-RAG, which its own Summary asserts as settled). L5 and L6 were the consequential ones: in both, the page *making* the contested claim was the page not carrying the objection. +- Fixes applied — staleness notes (2): [[2026-07-22-webinar-theses]] (synthesized from 7 sources, 10 now exist; thesis 3 "Skills are the new memory" is load-bearing and its counter-example was missing from its own tensions list; nothing from the last two ingests appears) and [[theo-konstantin-allie]] (predates Theo's second source and Thorsten; its closing "none of the three directly contradicts another" now misleads about the skills thread). Both preserved rather than rewritten, per the Update Policy's no-silent-large-rewrites rule. +- Open findings (not fixed — scope decisions): **(1)** "taste" is the vault's largest uncovered concept — 12 pages / 17 occurrences, one of four nouns in the through-line, the other three all have pages; named the meta-skill by Allie and the whole slop answer by Thorsten; recommend a concept page. **(2)** Query pages are a weakly-linked class (0/1/2/2 inbound); [[2026-07-22-webinar-theses]] is effectively orphaned despite being the most webinar-relevant page in the vault. **(3)** Watch items: [[2026-07-22-ai-is-stupid]] is thinly integrated (2 citing pages vs a median of ~17, plausibly correct since it restates rather than adds); [[emacsification-of-software]] + [[explosion-of-internal-software]] should merge if neither gains second-source support by the next lint. Considered and rejected as a new page: "trust calibration" — covered by [[make-more-cheap-code]] + [[code-as-throwaway]] + [[seniority-and-the-junior-squeeze]]; creating it would duplicate rather than consolidate. +- Contradiction inventory: 9 live disagreements, now all recorded on both sides. #1 (skills-as-memory vs no-skills-at-all) is the vault's most consequential open question because the webinar's central promise rests on it. +- Next: refresh [[2026-07-22-webinar-theses]] against all 10 sources (fixes the staleness and most of the weak-linking in one operation, and is the page closest to the deliverable); create the `taste` concept page; extend [[theo-konstantin-allie]] to four lenses. The falsification test flagged on [[skills-as-memory]] — same task with and without a skill, in a non-engineer's hands — remains the cheapest experiment that would move contradiction #1. + +## 2026-07-28 — query (webinar theses v2 — refresh) +- Intent: query +- Input: "Refresh 2026-07-22-webinar-theses" — acting on the top recommendation from [[2026-07-28-lint]] (v1 was two ingests stale and effectively orphaned). +- Pages created: [[2026-07-28-webinar-theses]] (17 theses from all 10 sources + the current deliverable state in `raw/notes/`). +- Pages updated: [[2026-07-22-webinar-theses]] (staleness note replaced with a **SUPERSEDED** banner summarising what changed and pointing forward — preserved as the 7-source state, not overwritten, per the Update Policy); backlinks added from [[explosion-of-internal-software]], [[skills-as-memory]], [[shedding-weight]] and [[2026-07-24-non-engineer-throwaway-verification]] so v2 does not repeat v1's orphaning; `index.md` (Queries section marks v1 superseded, v2 current). +- Method note: created as a **new dated page** rather than editing v1 in place. Query pages are dated Q&A snapshots (`wiki/queries/YYYY-MM-DD-.md`); rewriting the 07-22 file would have made its date lie and would have been a silent large rewrite of a dated artifact. +- Substantive changes to the thesis set: **(1) T3 reframed** from "skills are the new memory" to *"context you author beats context that's inferred"* — the v1 wording has a live counter-example in [[thorsten-ball]], and the reframe is what all four practitioners actually agree on (Konstantin's skills, Allie's foundation docs, Eugene's anti-memory position, Thorsten's `AGENTS.md` are all authored context). It survives all three readings logged on [[skills-as-memory]] and keeps the script's Memory→Skills stations intact. **(2) T5 upgraded from assertion to evidence** — [[explosion-of-internal-software]] supplies a non-engineer-shaped outside case (20-person club, phone, menu photo, ~2 hours), which defuses the "sure, *you* can do that, you're technical" objection against the talk's least-provable claim. **(3) Three new theses** the earlier set had no source for: T6a verification/ask-for-checks, T10 shedding weight ("which of your processes only exist because *you* were the bottleneck?"), T13a ask-for-15-options. **(4) T1 strengthened** — Thorsten's "stop tuning model choice; the dominant variable is the information you put in" makes harness-not-model the best-evidenced claim in the vault, and it is already the script's literal closing argument. **(5) New honest caveat** — token budget as a second, non-skill axis of the gap. +- Revised 30-min cut: T2 · T4 · T3 · T6a · T1 · T5 · T15, mapped to script beats. Changed from v1's recommendation: T3 reframed, T6a added, T12 promoted (it now has a teachable five-part prompt structure from Thorsten rather than an abstract principle), T9 demoted to Q&A. +- Also produced: a Q&A-ammo section stating the skills dissent honestly with the three readings and the presentation-safe framing, and **three gaps the refresh exposes in the current script** — no verification beat (the audience's first question is "can I trust it?" and the script never once shows the agent being checked), no "what stays yours" beat (the closing arc is entirely harness, not human), and the unsaid token-budget caveat against the "a laptop, one hour, and your real work" promise. +- Read directly for this refresh: `raw/notes/Webinar script.md` (current ladder: Chat box → ReAct → Tools → Memory → Skills → Process → OS), `raw/notes/Webinar Plan - From Chat Box to Your Own OS.md`, `raw/notes/my theses.md`. Notable: T4 and T1 are already dramatized in the script better than a slide could do it ("the notebook is tiny. On purpose."; "the model never changed"). +- Next: decide whether T6a earns a station or one line inside the Skills station (recommend the latter — a checker skill is one sentence of demo and costs no new level); decide whether T10 opens the talk rather than closing it; the remaining lint recommendations stand (create a `taste` concept page; extend [[theo-konstantin-allie]] to four lenses). The falsification test on [[skills-as-memory]] would settle T3's tension and would itself make a strong demo. + +## 2026-07-28 — maintenance (token-budget claim rescoped; webinar gap #3 withdrawn) +- Intent: maintenance +- Input: user objection to gap #3 in [[2026-07-28-webinar-theses]] — every LM vendor sells a subscription, that subscription covers even advanced users, so "budget" reduces to subscription cost, which is obvious. +- Assessment: **objection upheld, and the corpus supports it more strongly than the original write-up did.** The token-budget claim was recorded at ingest as an unqualified "the divide is also a *spending* gap" without weighing three counter-datapoints already in the vault: [[eugene]] runs 7 project-agents in parallel on a $200 plan; [[allie-miller]] runs ~100 agents and 36 workflows; and [[2026-07-14-sebastian-eugene-interview]] frames levelling as "a 20-year veteran and a fresh grad **on the same subscription**." No practitioner in the corpus reports a cost ceiling. The claim's real scope is **metered** pricing — [[amp]] sells usage, and Thorsten's pattern is parallel remote sandboxes and parked orbs ([[async-by-default]]) — which is a fleet cost, not a seat cost. +- Pages changed: [[enterprise-ai-reality]] (scoping sub-bullet added under the token-budget item, with the counter-evidence named), [[explosion-of-internal-software]] (the "two variables" bullet rescoped; its Contradictions entry partly resolved and marked tentative), [[context-as-scarce-resource]] ("context now has a price" → "…at fleet scale"; per-request scarcity restated as the binding constraint for individuals), [[overview]] (vault-level open question narrowed to fleet/enterprise allocation), [[2026-07-28-webinar-theses]] (gap #3 withdrawn with the reasoning recorded inline; T7's "second axis" line rescoped so the thesis stays on the skill gap). +- Claim preserved, not deleted, per rule 5 — Thorsten did say it and it stands in its own regime. What changed is scope and the counter-evidence, both now stated on every page carrying it. +- Residual open: metered/fleet pricing and enterprise budget allocation remain unanswered; [[eugene]]'s price-rise prediction ("what I now buy for 200 will cost about 1,000") would reopen the question for individuals if it holds — currently a forecast, not a constraint. Status: tentative. +- Notes: worth flagging as a process lesson — the claim came from a credible source and was written up the same session it arrived, without checking it against the vault's existing practitioner evidence. A source's framing of its own economics is not automatically the corpus's. + +## 2026-07-28 — query (verification beat design) +- Intent: query +- Input: "What are your suggestions for verification beat? What can we add?" — following the gap flagged in [[2026-07-28-webinar-theses]]. +- Pages created: [[2026-07-28-verification-beat-design]] (placement analysis, three options costed by seconds, drafted script copy in the script's voice, audience-translation lines, honest caveats). +- Pages updated: [[2026-07-28-webinar-theses]] (T6a now points to the design page); `index.md` (Queries). +- Key synthesis: **the verification beat and the "what stays yours" beat are the same beat** — verification is exactly where the human's remaining job lives ([[product-ownership]]) — so one insertion closes both gaps the refresh identified. Matters for a 30-min format. +- Placement argument: the audience's unease peaks at one specific existing line in the Process station ("I take my hands off the keyboard… Nobody is typing. It just... runs."). Answer it there, not in Q&A. Recommended split: introduce the checker skill at **Skills** (~60–90s), cash it in at **Process** (~20s, reuses an existing reveal), put the judgment half in the **closing arc** (~30s). +- Three options by cost: **(1)** ~15s and free — the Process station already reveals the prompt the shell wrote for its worker; add one visible self-check line to that artifact, no new demo steps. **(2)** ~60–90s recommended — a second skill whose only job is to check the first, demonstrated by breaking the state *by hand* so the audience sees the error before the agent reports it; introduces the producer/checker species split from [[2026-07-24-non-engineer-throwaway-verification]] at no new level. **(3)** ~30s standalone — **the script already contains a perfect ambiguity example**: the Process goal line says "keep the cube on the **right shelf**", which reads as *correct* shelf or *right-hand* shelf. The colon disambiguates it as written, so the unsafe version can be shown deliberately as a live demo of [[leave-less-room-for-imagination]] (currently thesis T12 with no demo) and as motivation for the checker. +- Closing half drafted to fill gap #2: a checker catches drift but can never catch a wrong rule ("if twenty degrees was the wrong number, both agents agree, confidently, forever") → "the machine checks whether the thing was done right; you check whether it was the right thing." Converts the ending from capability to the audience's own value. +- Caveats recorded rather than smoothed: a checker authored by the agent it checks is not independent verification (the same limit already logged on [[async-by-default]] — proof produced by the thing being checked is evidence, not verification); [[nina]]'s transcript-beats-summary finding is a standing counterweight against implying reading is now optional; and none of this has been tested on a non-engineer audience, which is why the audience-translation lines after option 2 are load-bearing rather than optional. +- Note: `raw/notes/Webinar script.md` was read but **not modified** (rule 1). All drafted copy lives in the wiki for the user to apply. +- Next: gap #2 is now addressed by this design; the remaining lint recommendations stand (create a `taste` concept page; extend [[theo-konstantin-allie]] to four lenses). Open decision from the refresh — whether T10 ("which processes only exist because you were the bottleneck?") opens the talk. + +## 2026-07-28 — maintenance (authorized raw edit: "right shelf" ambiguity demo) + theses legend +- Intent: maintenance +- Input: user — "That's a great idea about the 'right shelf'. would be a perfect demo. add it." Plus a question about the `T` shorthand. +- **Authorized raw edit** (exception to rule 1, on direct instruction; same pattern as the 2026-07-14 script edits logged above): `raw/notes/Webinar script.md`, Process station. Inserted the ambiguity demo between "…but a goal." and the real goal line — deliberately sloppy prompt ("keep the cube on the right shelf") → agent moves the cube to the right-*hand* shelf → "That's not what I meant. I meant the *correct* shelf. It heard: the shelf on the *right*." → "It didn't ask. It didn't hesitate… It just confidently did the wrong thing." → "Every gap you leave, it fills. And it fills it silently." → ties into the hands-off moment a minute later and reuses Marcus's own word: "Marcus said: don't let it drift. Turns out the first thing that drifts… is what I meant." → then the precise goal as originally written. +- Placement rationale: kept at Process rather than moved to Skills (the design page's alternative) because the user is adding the ambiguity demo *alone*, without the checker skill — standalone it is strongest where the ambiguous phrase already lives and where handing over control is imminent. If the checker skill (option 2 of [[2026-07-28-verification-beat-design]]) is added later, this beat should move earlier so the problem precedes its solution. +- **Stage-safety note added inline** (`_note:`): a modern model may disambiguate "right shelf" correctly from context, so this beat must be pinned to a deterministic response or a low-temp on-rails prompt. Consistent with the Plan's existing "never a naked live call" production rule. Without pinning, the demo can silently succeed and kill the point on stage. +- Effect on the wiki: this is the first *demo* of [[leave-less-room-for-imagination]] in the deliverable — thesis T12 previously had no dramatization. The script's own accidental ambiguity became the example. +- Pages changed: [[2026-07-28-webinar-theses]] — added a legend explaining the `T` shorthand (T = thesis; stable handles for cross-referencing; T1–T15 follow v1's order where the thesis survived; letter suffixes mark v2 additions placed beside their nearest relative instead of renumbering). This was an undocumented convention I introduced in v2 and the user was right to flag it. +- Next unchanged: decide on the checker skill (option 2) and the closing what-stays-yours half from [[2026-07-28-verification-beat-design]]; remaining lint recommendations stand (a `taste` concept page; extend [[theo-konstantin-allie]] to four lenses). + +## 2026-07-29 — ingest (А что если наВайб-Кодить / "What if we vibe-code it?") +- Intent: ingest +- Input: `raw/sources/А что если наВайб-Кодить.md` — viewer's conclusions from a 5:31 Russian YouTube video (author unknown; his company pays "millions a year" for Datadog). 11th source; `raw/sources/` fully ingested again. +- Pages created: [[2026-07-29-what-if-we-vibe-code-it]] (source), [[maintenance-is-the-real-cost]] (concept — 25th). +- Core of the source: **writing code was never the bottleneck — maintenance is.** Developers never skipped building their own Jira/Datadog for lack of ability; they skipped it because they didn't want to *run* the result. An internal service is a second IT business (bad for the company and the developer both); "I can write it in a week" ≠ "worth writing"; the vendor sells operational offload, not code. Evidence: the pendulum case — a company builds its own Jira clone (March 2026) and returns to a bought tracker, Linear, by July. Prescription: a build-vs-buy checklist (dependency size / ongoing support / operational load / second-business willingness). The author also retracts his own earlier "many services will die because of AI" claim. +- Why it matters to this vault: it is the **first dedicated counterweight to the build-everything-yourself thread** ([[explosion-of-internal-software]], [[emacsification-of-software]]) — and both of those pages had already flagged "maintenance is assumed away" as their own weakest point, so the objection was latent and is now sourced with the corpus's only *observed outcome* of the pattern (a reversal). Reconciliation recorded on both sides: Thorsten's club app *passes* the source's own checklist (tiny, personal, no SLA), his "teams will remix Riverside" prediction is what the checklist rejects — the disagreement is a threshold, not a winner. Separately, the source *agrees* with the vault's spine from a new angle: "writing was never the bottleneck" is the harness-over-model premise; the corpus now holds three named bottlenecks that don't compete — context (Thorsten, authoring time), verification (Theo, ship time), maintenance (this source, lifetime). +- Pages updated: [[explosion-of-internal-software]] (contradiction upgraded from self-criticism to sourced, threshold reconciliation, next-question sharpened), [[emacsification-of-software]] (maintenance objection sourced; `~/bin` fork passes, team remix doesn't), [[code-as-throwaway]] (new "lifetime boundary" bullet — throwaway is safe *because* unmaintained; first user converts code into a service), [[thorsten-ball]] (contradiction added, tentative both sides — one anecdote vs one prediction), [[overview]] (11 sources; frontier bullet counterweight; new divergence entry; concept count), [[2026-07-28-webinar-theses]] (T5 scoping note: the thesis survives — its examples are checklist-safe — and gains a one-sentence inoculation against the sharpest technical-audience pushback), `index.md`. +- Uncertainty flagged: the pendulum case is second-hand tweets with fuzzy company identification (tentative); the checklist is prescriptive, not observed; the author's own company is currently building a Datadog replacement — if it ships and survives, he becomes his own counterexample. Open question with no evidence either way in the corpus: does the maintenance objection survive *agents* doing the maintenance ([[agentic-loops]], [[async-by-default]])? +- Effect on the last lint's watch item: [[emacsification-of-software]] and [[explosion-of-internal-software]] were merge candidates "if neither gains second-source support by the next lint" — both now have second-source engagement (as a bounding counterpoint), which argues for keeping them separate with [[maintenance-is-the-real-cost]] as the shared boundary page. +- Next: the standing recommendations are unchanged (a `taste` concept page; extend [[theo-konstantin-allie]] to four lenses; the skills falsification test). New candidate question for the HR audience: which of their candidate tools (candidate knowledge base, transcribe→summarize) fall on the safe side of the build-vs-buy checklist — directly webinar-relevant if Q&A raises "should we build or buy?" diff --git a/raw/assets/Dan Martell - Scale Your Business Workbook.pdf b/raw/assets/Dan Martell - Scale Your Business Workbook.pdf deleted file mode 100644 index 0eadf2f..0000000 Binary files a/raw/assets/Dan Martell - Scale Your Business Workbook.pdf and /dev/null differ diff --git a/raw/notes/Webinar script.md b/raw/notes/Webinar script.md index e8d6824..22e724a 100644 --- a/raw/notes/Webinar script.md +++ b/raw/notes/Webinar script.md @@ -424,6 +424,42 @@ Watch what happens when I give it — not a task... ...but a goal. +Actually — hold on. + +Before I hand over control, let me be sloppy on purpose. + +"keep the cube on the right shelf" + +{AI moves the cube to the right-hand shelf} + +_note: stage-critical — a modern model may well disambiguate "right shelf" correctly from context. Pin this beat to a deterministic response (or a low-temp on-rails prompt) so it reliably picks the right-hand shelf._ + +That's not what I meant. + +I meant the *correct* shelf. + +It heard: the shelf on the *right*. + +And notice what it didn't do. + +It didn't ask. It didn't hesitate. It didn't flag anything. + +It just confidently did the wrong thing. + +Every gap you leave, it fills. + +And it fills it silently. + +Right now that's harmless — I'm sitting here, I can see the cube. + +But in a minute I'm going to walk away from this keyboard. + +Marcus said: don't let it drift. + +Turns out the first thing that drifts... is what I meant. + +So let's say exactly what we mean. + "keep the cube on the right shelf: below 20 — top, above 20 — bottom. continuously." {AI spawns a process — "started process 1"} diff --git a/raw/sources/А что если наВайб-Кодить.md b/raw/sources/А что если наВайб-Кодить.md new file mode 100644 index 0000000..95929b3 --- /dev/null +++ b/raw/sources/А что если наВайб-Кодить.md @@ -0,0 +1,107 @@ +# Выводы по видео + +**Источник:** https://www.youtube.com/watch?v=zBcWcignqng +**Название:** А что если наВайб-Кодить? +**Длительность:** 5:31 + +--- + +## Главный тезис + +**Нейросети открыли ящик Пандоры: теперь любой сервис можно быстро переписать «под себя» — и в этом главная ловушка.** Проблема современного софта никогда не заключалась в написании кода. Проблема — в его поддержке. Разработчики не писали свои аналоги Jira, Datadog и т.д. не потому что *не могли*, а потому что *не хотели*. И зря забывают об этом сейчас, вдохновившись возможностями ИИ. + +--- + +## Иллюстрация: маятник «сделали своё → вернулись к покупному» + +Автор приводит два твита с разницей в несколько месяцев: + +| Дата | Что произошло | +|---|---| +| Март 2026 | Компания сделала свой аналог Jira со всем нужным функционалом и переехала на него | +| Июль 2026 | Та же (или похожая) компания вернулась к покупке трекера (Linear), потому что не захотели тащить свой продукт | + +Это типичный сценарий эпохи вайб-кодинга: собрать за пару недель — легко, тащить дальше — невозможно. + +--- + +## Почему раньше не переписывали всё сами + +Общее заблуждение: «раньше не могли, а теперь с ИИ смогли». **Это неправда.** + +- Разработчики всегда могли написать любой сервис — руками, командой, за несколько месяцев. +- Не писали по одной причине: **не хотели управлять этим сервисом дальше**. +- Сам код — не проблема. Проблема начинается после первого пользователя. + +--- + +## Что ломается, как только у продукта появляются пользователи + +Как только сервис живёт и масштабируется, на разработчиков сваливается: + +- баги и регрессии; +- запросы на новые фичи; +- «что-то не так работает / не там работает / не сработало»; +- логи, мониторинг, дежурства; +- ответственность за аптайм. + +Проект, который делался «чтобы сэкономить на подписке», превращается в **отдельную постоянную работу** с выделенными людьми и временем — ровно то, что делала компания-вендор, которой вы платили. + +--- + +## Ключевой парадокс: два бизнеса вместо одного + +Если основной бизнес компании — например, «условный ChatGPT», а в фоне она тащит свой self-hosted трекер / логгер / что-то ещё, то она: + +- либо **переходит из одного бизнеса во второй**, +- либо **совмещает два IT-бизнеса в одном**. + +Плохо для всех: + +| Кому плохо | Почему | +|---|---| +| Компании | Платит за один продукт, а команда пилит второй | +| Разработчику | Есть основная работа (ругают, если не сделал) + второстепенная (ругают, если не сделал) | + +--- + +## Личный кейс автора + +- В его компании используют **Datadog**. +- Платят «буквально миллионы в год» за работу с логами и их хранение. +- Хотят заменить и уже разрабатывают свой аналог + присматриваются к более дешёвым альтернативам. +- Признаёт: в предыдущем ролике сказал, что «многие сервисы умрут из-за ИИ» — и **был неправ**. + +--- + +## Сквозные принципы + +1. **Написание кода — не бутылочное горлышко.** Никогда не было. +2. **Стоимость софта = стоимость поддержки**, а не разработки. +3. **«Могу написать за неделю» ≠ «стоит писать».** Между этими двумя утверждениями — годы саппорта. +4. **Внутренний сервис — это внутренний бизнес.** Со своими SLA, дежурствами, багфиксами, roadmap. +5. Вендор берёт деньги не за код, а за то, что снимает с вас операционную нагрузку. + +--- + +## Что делать: практический вывод автора + +> «Не пытайтесь переписать всё.» + +Оценивать замену сторонних решений стоит по чек-листу: + +- **Размер зависимости.** Небольшие библиотеки без развития — можно переписать. +- **Требует ли дальнейшей поддержки?** Если да — считайте это отдельным проектом. +- **Какую операционную нагрузку добавит?** Мониторинг, багфикс, дежурства, дев-время. +- **Готов ли бизнес открывать второй IT-бизнес внутри себя?** + +Если ответы «да / много / нет» — оставайтесь на платном сервисе, даже имея под рукой Claude / Antigravity / Codex. + +--- + +## Кому это полезно + +- **Тимлидам и техлидам,** которые под впечатлением от вайб-кодинга собираются «за спринт заменить Jira / Datadog / Sentry». +- **Основателям стартапов,** решающим build vs buy для инфраструктурных инструментов. +- **Fullstack-разработчикам,** прикидывающим себестоимость «своего маленького SaaS-клона». +- **Инженерам,** оценивающим ROI миграции с внешнего сервиса на in-house решение. diff --git a/wiki/comparisons/theo-konstantin-allie.md b/wiki/comparisons/theo-konstantin-allie.md index 5bc430d..4b2ac50 100644 --- a/wiki/comparisons/theo-konstantin-allie.md +++ b/wiki/comparisons/theo-konstantin-allie.md @@ -2,6 +2,12 @@ #comparison +> **Staleness note (added 2026-07-28 by [[2026-07-28-lint]]).** Written 2026-07-16 from one source per speaker. Two developments since, neither reflected below: +> - **[[theo-browne]] has a second source** ([[2026-07-24-youre-reading-way-too-much-code]]). The "On code" row and the "deskilling vs re-skilling" tension both read his position as bare disposability; he has since supplied its *discipline* ([[make-more-cheap-code]] — keep hand-verification of what ships, generate 100×+ more that never ships) and explicitly disowned shipping unreviewed slop. +> - **The closing claim "None of the three directly contradicts another; the disagreements in this corpus are elsewhere" is now materially incomplete.** [[thorsten-ball]] contradicts the skills thread this page calls "the vault's strongest cross-source thread" — he uses no skills, no MCP, no slash commands. He is outside this page's three-way scope, but a reader taking the closing line at face value would conclude the skills convergence is uncontested. It is not; see [[skills-as-memory]]. +> +> Content below is preserved as the 2026-07-16 state. Recommended refresh: extend to four lenses, or add Thorsten as an explicit dissent column. + ## Summary Three speakers describe the **same underlying change** — models now improve faster than people can, and the durable advantage has moved from the model to the *system you wrap around it* — but from three non-overlapping vantage points: diff --git a/wiki/concepts/async-by-default.md b/wiki/concepts/async-by-default.md new file mode 100644 index 0000000..4b79e3f --- /dev/null +++ b/wiki/concepts/async-by-default.md @@ -0,0 +1,40 @@ +# Async by Default (Orbs and Proof) + +#concept + +## Summary + +If agent work takes sixteen minutes and you are doing something else, latency stops being a cost. [[thorsten-ball]]'s working mode: delegate into a **remote sandbox**, walk away, run several in parallel, and — since you are waiting anyway — **ask the agent for proof** rather than a claim of success. + +## Current Understanding + +**The orb.** [[amp]]'s unit of work is a remote sandbox tied to one conversation. It sleeps when idle and wakes on typing; it streams to phone, laptop and TUI as the same conversation; and **one URL packages the thread + the agent + the computation + the diff**. Share the URL and a teammate opens the orb and takes over. Agent-to-agent messaging turns this multiplayer: "I found another bug" → "launch another orb to fix it" → new checkout, new branch, new agent, in parallel. + +**The old objections collapsed.** Cloud IDEs (Cloud9 and friends) died on latency, key bindings, "I can't SSH in," and missing language servers. Thorsten's rebuttal: *who cares about latency when you're waiting for tokens per second anyway?* — and nobody uses editors, key bindings or language servers the way they did when those objections were formed. The objection stack was about a workflow that no longer exists. + +**Ask for proof.** Quinn (AMP's CEO): *"You're async anyway — so ask the agent to give you proof."* Screenshots, benchmarks, dark-mode *and* light-mode variants, fifty tests in parallel. This is the delegation-side counterpart to [[make-more-cheap-code]]: cheap generated artifacts exist to make a claim checkable, and asking for three of them costs you nothing when you are not sitting there watching. In practice at AMP: screenshot a bug → send it → an orb returns a fix → spot check → merge; the designer "never fixed so many paper cuts." + +**The prediction.** Local dev effort goes away, replaced by remote sandboxes — with the caveat that 15+ sandbox providers are already racing margins to zero, which Thorsten himself calls unsustainable. + +## Evidence + +- Orbs, sleep/wake, one-URL packaging, multiplayer handoff, the 16-minute live demo, the collapsed cloud-IDE objections, Quinn's proof line, paper-cut velocity, local-dev prediction, infra-margin prediction — [[2026-07-28-agentic-engineering-10x-developer]]. +- Fire-and-forget as a native harness mode; completion notifications as what makes background agents usable — [[2026-07-14-skills-based-on-git]], [[harness]]. +- Scheduled agents producing while you sleep (the non-engineer version) — [[personal-ai-operating-system]]. + +## Related Pages + +- Concepts: [[harness]] (async is one of its two modes), [[agentic-loops]], [[make-more-cheap-code]] (proof artifacts are throwaway code with a job), [[shedding-weight]] (async is what makes killing the backlog possible — parked agents replace queued tickets), [[personal-ai-operating-system]], [[context-as-scarce-resource]] +- Entities: [[thorsten-ball]], [[amp]], [[claude-code]] + +## Contradictions / Uncertainty + +- **Attention, not latency, is the real budget.** Five parallel orbs produce five diffs that a human must still review; [[make-more-cheap-code]] argues reading is the scarce resource. Async multiplies generation without multiplying review capacity, and the source does not address the pile-up. Status: tentative — this is the same open question logged on [[make-more-cheap-code]] about reviewing *agent behaviour* becoming the new attention sink. +- **Proof is produced by the thing being checked.** A screenshot from the agent that made the change is evidence, not verification; the failure mode where an agent produces a convincing artifact of work it did not do is unaddressed. +- **Remote sandboxes vs compliance.** Code and conversation in a vendor's cloud is exactly what [[enterprise-ai-reality|locked-down enterprises]] forbid. Also sits against [[eugene]]'s consolidated *local* workspace pitch ([[harness]]) — though the two are compatible if the consolidation point is the interface rather than the compute. +- "Local dev is going away" comes from a company selling remote sandboxes. Status: tentative. + +## Next Questions + +- What is the non-engineer's orb? The corpus has scheduled workflows (Allie) and completion notifications (Eugene) but nothing that packages a resumable, shareable unit of work for a non-technical user. +- Which proofs actually catch drift? If [[leave-less-room-for-imagination|the damage is what you don't notice]], a screenshot proves the happy path and nothing else — the proof list needs a design, not just a habit. diff --git a/wiki/concepts/build-for-the-agent-not-the-human.md b/wiki/concepts/build-for-the-agent-not-the-human.md new file mode 100644 index 0000000..1dd6ceb --- /dev/null +++ b/wiki/concepts/build-for-the-agent-not-the-human.md @@ -0,0 +1,37 @@ +# Build for the Agent, Not the Human + +#concept + +## Summary + +A product philosophy from [[thorsten-ball]]: if you start something on the frontier today, **no human should have to fill out a form.** Anything a human can do on your site, they should be able to have an agent do — and ideally they should **bring their own agent**, because "nobody wants to use your shitty built-in agent." + +## Current Understanding + +- **The admin panel that dies.** Building a food-ordering app from a photo of a menu, the agent also produced an admin UI for editing prices and spellings. Thorsten's reaction: *"I'm never going to open that. I'll just send another photo and say 'fix the pricing.'"* The insight underneath it is general: **a lot of admin UI existed only so that no code had to change.** Once changing code is cheap, the UI layer built to avoid changing code is pure [[shedding-weight|weight]]. +- **The same for content dashboards.** WordPress-style admin: "here's my draft, add this header image, publish, spell-check" — one sentence instead of a session of clicking. +- **Bring your own agent** is the sharp part, and it cuts against most 2026 product roadmaps: the differentiator stops being *your* assistant and becomes whether your surface is drivable by *the user's* assistant. That makes agent-accessibility a product feature rather than an integration checkbox. +- **Consequence for moats.** If every surface is agent-drivable and every agent can remix software ([[emacsification-of-software]]), general-purpose SaaS loses the lock-in that UI familiarity used to provide. Thorsten's own prediction list says it plainly: it is unclear what software survives. + +**Read carefully, this is not "no UI."** The claim is that UI built as a *substitute for changing the system* dies, and UI built as a genuinely better interface survives. The distinction matters for the webinar's [[personal-ai-operating-system|OS framing]], where the endpoint is a *smaller* interface (a button that already knows what the email said) rather than no interface — arrived at from the opposite direction: Thorsten deletes UI so he can prompt, the OS framing builds tiny UI so you need not prompt. Both are the same underlying claim that the generic chat box and the generic admin panel are the two things being squeezed out. + +## Evidence + +- "No human should have to fill out forms," bring-your-own-agent, the food-app admin panel, the WordPress example — [[2026-07-28-agentic-engineering-10x-developer]]. +- Erosion of software moats via remixability (prediction 3 in the same source). + +## Related Pages + +- Concepts: [[shedding-weight]] (the parent move), [[emacsification-of-software]], [[explosion-of-internal-software]], [[personal-ai-operating-system]] (the interface question from the user's side), [[async-by-default]] +- Entities: [[thorsten-ball]], [[amp]] + +## Contradictions / Uncertainty + +- **Who operates the software if forms die?** Thorsten's answer is "prompt the agent" — which assumes exactly the prompting competence the corpus's HR interviews identify as the real bottleneck ([[levels-of-ai-usage]], and Nina/Yulia's *friction, not resistance* finding). An admin panel is a poor interface for an expert and a good one for a beginner. Status: tentative. +- "Bring your own agent" is asserted by someone who sells an agent; the business model that survives universal BYOA is not addressed. +- No account of authorization: an agent-drivable surface is also an agent-*abusable* surface, and the source says nothing about permissions, rate limits, or attribution. + +## Next Questions + +- What is the minimum an existing product must expose to be genuinely agent-drivable — an API, an `AGENTS.md`, structured error messages, or something else? +- Does the webinar audience want fewer forms or *better* forms? Worth asking directly, since it decides whether the OS pitch lands as liberation or as loss of a familiar surface. diff --git a/wiki/concepts/code-as-throwaway.md b/wiki/concepts/code-as-throwaway.md index 7d3f3b4..7de8a94 100644 --- a/wiki/concepts/code-as-throwaway.md +++ b/wiki/concepts/code-as-throwaway.md @@ -16,6 +16,11 @@ When the cost of writing code trends to zero, code stops being a precious asset. - **The trust carve-out.** Eugene puts a date and a boundary on it: "Code isn't something elite anymore. From 4.6 on, the code is safe enough — though **authorization and payments** I still wouldn't trust to Claude." Cheap code does not mean uniformly trusted code; the exceptions are where a silent error is unrecoverable rather than merely wrong. Consistent with the safety-critical exception noted below. +- **The production datapoint.** [[thorsten-ball]] reports **99% of [[amp]]'s code is written by AI** — the corpus's only figure from inside a shipping company rather than an individual workflow, and the strongest available answer to "does this survive contact with a real product?" He polled his team offering 99%+, 90–99% and <90%; the one engineer who said he still wrote "a bunch" by hand landed at ~95% when pushed. Self-reported, and from a company that sells an agent — but specific. +- **Slop is a human problem.** "Most of slop comes from humans not having good product. With AI they can just build trash products faster." Slop = lack of ideas, lack of playfulness, not knowing what you want to exist — *not* an AI defect. His counter-demonstration is taste at AI speed: 15 generated icon variants across styles and 18 palettes, one picked by hand. This converges with Theo's anti-slop stance from a different direction — Theo defends cheap code with *verification discipline*, Thorsten with *taste* — and both reject the vibe-coder reading of this page. + +- **The lifetime boundary.** [[2026-07-29-what-if-we-vibe-code-it]] adds the third cost besides writing and verifying: **maintenance**. Throwaway code is safe *because it is never maintained* — the trap begins at the first user, which converts code into a service ([[maintenance-is-the-real-cost]]). "I can write it in a week ≠ it's worth writing" is Theo's ship/no-ship line restated over the artifact's lifetime rather than at review time. + Caveat: legacy/hobby niches persist (COBOL in banks — no training data; coding "for the love of it, like an old-timer car") — but not where time, quality, and money matter. ## Evidence @@ -24,15 +29,18 @@ Caveat: legacy/hobby niches persist (COBOL in banks — no training data; coding - Kill code without guilt, guilt-merging, G-brain markdown tier — [[2026-07-14-everything-we-knew-about-software-has-changed]]. - "Code isn't elite anymore" from 4.6 on; authorization and payments withheld; browser-over-emulator testing note — [[2026-07-21-larysa-interview]]. - Ship/no-ship line, four tiers, 100-lines-of-slop-per-shipped-line, "make more cheap code" — [[2026-07-24-youre-reading-way-too-much-code]]. +- 99% AI-written at AMP; the hand-coding poll; "slop comes from humans"; the 15-icon-variant workflow — [[2026-07-28-agentic-engineering-10x-developer]]. +- Writing was never the bottleneck; cost of software = maintenance; "can write in a week ≠ worth writing" — [[2026-07-29-what-if-we-vibe-code-it]]. ## Related Pages -- Concepts: [[make-more-cheap-code]], [[product-ownership]], [[think-wider-not-bigger]], [[skills-as-memory]], [[decoupling-identity-from-profession]], [[leave-less-room-for-imagination]] -- Entities: [[theo-browne]], [[sebastian]], [[eugene]] +- Concepts: [[make-more-cheap-code]], [[maintenance-is-the-real-cost]] (the lifetime boundary), [[product-ownership]], [[think-wider-not-bigger]], [[skills-as-memory]], [[decoupling-identity-from-profession]], [[leave-less-room-for-imagination]], [[emacsification-of-software]] (cheap code makes the bespoke fork rational), [[explosion-of-internal-software]], [[shedding-weight]] +- Entities: [[theo-browne]], [[sebastian]], [[eugene]], [[thorsten-ball]] ## Contradictions / Uncertainty - "Most code isn't high-value" is a generalization; safety-critical/regulated code is a clear exception (see [[enterprise-ai-reality]]). +- The 99% figure is self-reported by a founding engineer at the company selling the agent, and describes a codebase whose authors are all expert users of that agent. It bounds what is *possible*, not what is typical. Status: tentative. ## Next Questions diff --git a/wiki/concepts/context-as-scarce-resource.md b/wiki/concepts/context-as-scarce-resource.md index 0923213..2972190 100644 --- a/wiki/concepts/context-as-scarce-resource.md +++ b/wiki/concepts/context-as-scarce-resource.md @@ -18,6 +18,10 @@ Context pressure explains several otherwise-separate design choices: The human role has climbed prompt-engineer → **context-engineer** → harness-builder → loop-engineer, tracking exactly this concern. +**Information beats tuning** ([[2026-07-28-agentic-engineering-10x-developer]]). [[thorsten-ball]] states the strongest version: once you have a frontier model, **the dominant variable in output quality is the information you put in** — not which model, and not the effort level (medium vs high vs ultra). "If you're mad your model doesn't use camelCase, rethink your software engineering, not the model." He names the agent's only two information sources — **training data** (a senior engineer who's seen it all, but lossy and possibly stale) and **the context window** (your prompt, plus whatever the codebase and `AGENTS.md` supply) — and the operative asymmetry: a model cannot turn a thin prompt into a good one. Note what he does *not* conclude: the fix is a better-tended codebase and a longer prompt, not [[skills-as-memory|skills]] (contested there). + +**Context now has a price — at fleet scale.** The same source names **token budget** as one of two variables separating winners from losers, alongside knowing how to use agents. Context has always been scarce per-request; this is the corpus's first claim that it is also scarce per-*wallet*. Scope, corrected 2026-07-28: the claim comes from **metered** usage (parallel remote sandboxes), and under a flat consumer subscription the corpus's own heavy users report no ceiling — so per-request scarcity remains the binding constraint for individuals, and per-wallet scarcity is a fleet and enterprise concern. See [[enterprise-ai-reality]] and [[explosion-of-internal-software]]. + **The supply-side facet** ([[2026-07-22-ai-is-stupid]]): before context is *scarce* it is usually *absent*. "Intelligence without context loses to context without intelligence" — ten Nobel laureates asked about your sales month can only cite industry averages, while your rank-and-file employee answers better because they see your funnel, clients, and deals. The default "stupid AI" experience is a strong model given neither business context nor a [[harness]]; the fix is investing in context infrastructure (data, memory, integrations) before reaching for a bigger model. ## Evidence @@ -25,11 +29,13 @@ The human role has climbed prompt-engineer → **context-engineer** → harness- - Smart zone, summarization decay, "context is the most valuable resource," tool/skill loading mechanics — [[2026-07-14-skills-based-on-git]]. - Context engineering vs prompt engineering; foundation docs as durable context — [[2026-07-14-gap-between-ai-users-irreversible]]. - "Intelligence without context loses"; Nobel-vs-employee analogy; invest in context before model upgrades — [[2026-07-22-ai-is-stupid]]. +- Information > model choice > effort level; the two information sources; token budget as a winner/loser variable — [[2026-07-28-agentic-engineering-10x-developer]]. +- Reading costs attention — the human-side analog of the same scarcity — [[make-more-cheap-code]], [[2026-07-24-youre-reading-way-too-much-code]]. ## Related Pages -- Concepts: [[harness]], [[skills-as-memory]], [[agentic-loops]], [[evolution-of-agent-tooling]], [[personal-ai-operating-system]] -- Entities: [[konstantin]], [[allie-miller]] +- Concepts: [[harness]], [[skills-as-memory]], [[agentic-loops]], [[evolution-of-agent-tooling]], [[personal-ai-operating-system]], [[make-more-cheap-code]], [[explosion-of-internal-software]], [[enterprise-ai-reality]] +- Entities: [[konstantin]], [[allie-miller]], [[thorsten-ball]] ## Contradictions / Uncertainty diff --git a/wiki/concepts/emacsification-of-software.md b/wiki/concepts/emacsification-of-software.md new file mode 100644 index 0000000..e598625 --- /dev/null +++ b/wiki/concepts/emacsification-of-software.md @@ -0,0 +1,38 @@ +# The Emacsification of Software + +#concept + +## Summary + +Emacs users have always forked plugins, rewritten them for their own config, and never contributed back. [[thorsten-ball]] (citing a blog post of this name) argues **that behaviour is now becoming the default for all software**: when an agent can bend someone else's program to your exact needs in two minutes, the bespoke fork beats the upstream contribution. + +## Current Understanding + +- **The worked example:** he forked a diff viewer called **hunk**, pointed [[amp]] at it, and said "add Gruvbox dark hard theme, add file-checkoff in the sidebar, compile, drop it in `~/bin`." Two minutes of agent time. **No reason to upstream** — the change is bespoke to him, and the cost of maintaining a personal fork has collapsed along with the cost of writing the patch. +- **The blast radius expands.** Not just individuals with small tools: he expects teams and companies to remix mid-sized software. His examples: "I want Riverside but audio-only," or video-only. +- **Why it matters commercially:** general-purpose SaaS has historically been defended by the gap between "close enough" and "exactly what I want." Agents close that gap for free, which is why his prediction list includes *it is unclear what software survives* — remixability plus per-user custom versions erode the moat. +- **The OSS side effect:** contributions dry up where the incentive to upstream was mostly "so I don't have to maintain a fork." This compounds a claim already in the corpus — that open-source contribution graphs are worth almost nothing now, and giving away near-free code costs little ([[code-as-throwaway]], via [[sebastian]]). + +**Relationship to the sibling concept.** [[explosion-of-internal-software]] is about building tools that never existed; this page is about *remixing tools that do*. Same shift — software becoming personal rather than general — from opposite starting points, and both feed the webinar's "little tools you make for yourself" thesis. + +## Evidence + +- The Emacsification framing, the hunk fork, the Riverside remix prediction, expanding blast radius — [[2026-07-28-agentic-engineering-10x-developer]]. +- OSS growth as low-cost giveaway, contribution graphs devalued — [[2026-07-14-sebastian-eugene-interview]]. +- Cost-of-code → zero as the enabling condition — [[code-as-throwaway]]. + +## Related Pages + +- Concepts: [[explosion-of-internal-software]] (sibling mechanism), [[maintenance-is-the-real-cost]] (the bounding counterweight), [[code-as-throwaway]] (the enabling economics), [[build-for-the-agent-not-the-human]] (what happens to the products being remixed), [[shedding-weight]], [[make-more-cheap-code]] (a fork nobody else sees is tier-A/B code with a long life) +- Entities: [[thorsten-ball]], [[amp]], [[sebastian]] + +## Contradictions / Uncertainty + +- **Maintenance is assumed away.** A two-minute fork is cheap; a fork carried across three years of upstream security patches is not. The source does not address rebasing, CVEs in the parent project, or what happens when the agent that built the fork can no longer reconstruct it. *(Sourced 2026-07-29:* [[2026-07-29-what-if-we-vibe-code-it]] *makes exactly this objection — [[maintenance-is-the-real-cost]] — and its pendulum case (in-house Jira clone abandoned for Linear within four months) is the mid-size "Riverside but audio-only" prediction failing in the wild. The personal `~/bin` fork still passes that source's checklist; the team/company remix he predicts does not.)* +- **Who maintains the upstream** if the people capable of patching it now all fork silently? The prediction is stated as an observation, with no answer for the commons problem it describes. +- Untested against [[enterprise-ai-reality]]: a bespoke unaudited fork in `~/bin` is precisely what locked-down corporate environments forbid. + +## Next Questions + +- Does a personal fork count as a durable artifact, or is it disposable in the [[code-as-throwaway]] sense — regenerated from a prompt against a fresh upstream each time you need it? The second reading is more consistent with the rest of the corpus and would dissolve the maintenance objection. +- Is there a non-engineer version of this — remixing a tool you use rather than one you can compile? diff --git a/wiki/concepts/enterprise-ai-reality.md b/wiki/concepts/enterprise-ai-reality.md index fd8cc0c..5123f0a 100644 --- a/wiki/concepts/enterprise-ai-reality.md +++ b/wiki/concepts/enterprise-ai-reality.md @@ -12,22 +12,28 @@ The indie/practitioner world and the regulated-enterprise world diverge sharply. - **The business opportunity:** *scalable, manageable, company-standard harnesses for larger engineering teams.* The gap between what individuals can do (custom [[harness]]) and what enterprises can allow **is** the product. - **Governance vs leverage tension:** individuals get maximum leverage from personal harnesses ([[eugene]]); enterprises must standardize and control ([[sebastian]]). Unresolved — and monetizable. - **Adjacent constraints:** the [[seniority-and-the-junior-squeeze|"read what you approve"]] security concern is amplified at scale; safety-critical/regulated code is the clear exception to [[code-as-throwaway|"most code isn't high-value"]]. +- **A second divide: the token budget** (added 2026-07-28; **scope corrected 2026-07-28** — see below). [[thorsten-ball]] names two variables separating winners from losers — knowing how to use agents, and **having the token budget to do it**. It cuts both ways for this page: an enterprise can buy budget an individual cannot, while a locked-down enterprise may withhold it from the people who would use it best. Whoever controls the budget controls how far [[explosion-of-internal-software|internal software]] spreads. Thorsten names the variable and says nothing about who pays. + - **Scoping correction.** This was first written here as "the divide is also a *spending* gap," which overstates it. Thorsten's pricing regime is **metered**: [[amp]] sells usage, and his working pattern is parallel remote sandboxes and parked orbs ([[async-by-default]]) — a fleet cost, not a seat cost. Under a **flat consumer subscription** the corpus's own evidence points the other way: [[eugene]] runs 7 project-agents in parallel on a $200 plan, [[allie-miller]] runs ~100 agents and 36 workflows, and neither reports hitting a cost ceiling — while [[2026-07-14-sebastian-eugene-interview]] frames levelling as "a 20-year veteran and a fresh grad **on the same subscription**." For individual and small-team use the budget is one subscription; the token-budget variable bites at fleet scale and under metered pricing, which is where Thorsten sits and where enterprises will land. +- **The frontier's advice does not transfer.** [[shedding-weight]] — kill the backlog, kill CI that repeats the agent's tests, kill local dev in favour of remote sandboxes ([[async-by-default]]) — describes a startup that owns its own process. In a regulated shop the pipeline, the audit trail and the ticket history frequently *are* the deliverable to a regulator, and code sitting in a vendor's remote sandbox is precisely what Sebastian's clients forbid. The gap between what the frontier recommends and what compliance permits is the same gap this page calls the market. ## Evidence - Managed VMs / zero self-install, Roche ~1,200 engineers, banks banned→adopting, "company-managed resource," "the interesting market" — [[2026-07-14-sebastian-eugene-interview]]. +- Token budget as a winner/loser variable; the frontier playbook (kill backlog/CI/local dev, remote sandboxes) that compliance cannot follow — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages -- Concepts: [[harness]], [[seniority-and-the-junior-squeeze]], [[code-as-throwaway]] -- Entities: [[sebastian]], [[virtido]], [[eugene]] -- Tools: [[claude-code]] +- Concepts: [[harness]], [[seniority-and-the-junior-squeeze]], [[code-as-throwaway]], [[shedding-weight]], [[async-by-default]], [[explosion-of-internal-software]], [[context-as-scarce-resource]] +- Entities: [[sebastian]], [[virtido]], [[eugene]], [[thorsten-ball]] +- Tools: [[claude-code]], [[amp]] ## Contradictions / Uncertainty - How AI transforms *huge* (~1,200-engineer, multi-year) programs is explicitly unknown even to Sebastian. - Whether [[virtido|Virtido]] itself is building the company-managed harness, or just naming the market, is unstated. +- [[amp]]'s orb model (code, conversation and diff living in a vendor's remote sandbox) is a direct test case for this page and the source never addresses it. Whether the frontier's unit of work is adoptable at all under compliance is open. Status: tentative. ## Next Questions - What is the minimal compliant feature set for a centrally-managed enterprise harness? +- Who controls the token budget in a large organisation, and is it allocated by role, by team, or by request? The corpus has no evidence either way, and it decides who actually gets to use the tools. diff --git a/wiki/concepts/evolution-of-agent-tooling.md b/wiki/concepts/evolution-of-agent-tooling.md index 4e8f104..c85dda6 100644 --- a/wiki/concepts/evolution-of-agent-tooling.md +++ b/wiki/concepts/evolution-of-agent-tooling.md @@ -16,21 +16,25 @@ Konstantin's three-generation map of how agents get capabilities: **Tools (2022 **When to use which:** Skills when tasks are unknown/diverse or tool count is ~5–50; MCP when the agent is narrow, tasks are uniform, and the same small toolset applies every time. The line blurs — Claude Code converts MCP servers *into* skills (file laid down, functions not all injected), erasing most MCP downsides. Konstantin doesn't hate MCP; its problems are largely solved. +**A fourth position: skip the progression.** [[thorsten-ball]] uses none of the three generations as user-authored artifacts — no skills, no MCP servers, no slash commands — and locates capability instead in the *codebase* plus `AGENTS.md` plus a rich prompt ([[2026-07-28-agentic-engineering-10x-developer]]). [[amp]] does ship structure (Oracle/Painter/Puck sub-agents, a model/effort dial), but the vendor curates it, not the user. Read against this table, his claim is that the progression's real axis was never tools → MCP → skills but **who supplies the context and where it lives** — and that for someone working in one well-tended repo, the repo wins. See the three competing readings logged on [[skills-as-memory]]. + ## Evidence - Three generations, per-generation problems, skills-vs-MCP decision table, Claude-Code-turns-MCP-into-skills caveat — [[2026-07-14-skills-based-on-git]]. - Skills as portable markdown folders across Claude/Perplexity/Gemini — [[2026-07-14-gap-between-ai-users-irreversible]]. +- The skip-it-all position; vendor-curated sub-agents in place of user-authored skills — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages - Concepts: [[skills-as-memory]], [[harness]], [[context-as-scarce-resource]] -- Tools: [[claude-code]], [[hermes]] -- Entity: [[konstantin]] +- Tools: [[claude-code]], [[hermes]], [[amp]] +- Entity: [[konstantin]], [[thorsten-ball]] ## Contradictions / Uncertainty - "Everything changes every 6 months" — MCP was just ratified and A2A is already wanted; this map may shift quickly. Status: tentative. +- The map assumes each generation *supersedes* the last. Thorsten's practice suggests a parallel track that never enters the table at all (repo + `AGENTS.md`), which would make the progression a history of *one* branch rather than of agent capability as such. Status: tentative. ## Next Questions -- Where does agent-to-agent (A2A) sit in this progression? +- Where does agent-to-agent (A2A) sit in this progression? *(Partial datapoint 2026-07-28: [[amp]] shipped agent-to-agent messaging and a meta-agent that spawns and controls other agents — see [[async-by-default]]. It arrived as a harness feature, not as a fourth generation of capability-delivery, which the table would not have predicted.)* diff --git a/wiki/concepts/explosion-of-internal-software.md b/wiki/concepts/explosion-of-internal-software.md new file mode 100644 index 0000000..0d7fe2e --- /dev/null +++ b/wiki/concepts/explosion-of-internal-software.md @@ -0,0 +1,38 @@ +# Explosion of Internal Software + +#concept + +## Summary + +The layer of organisational life that used to be **one Excel file, one wiki page, and one hacky script** is about to be replaced by actual software, because building it now costs an evening instead of a quarter. [[thorsten-ball]] encoded his 20-person club's ordering process in **~2 hours of typing on his phone**. This is the corpus's strongest external validation of the webinar's thesis — *little tools you make for yourself*. + +## Current Understanding + +- **The before/after.** Before, internal software was whatever survived the cost-benefit test against a spreadsheet — which almost nothing did. Now the test is trivially passed, so the Excel/wiki/hack layer gets replaced by real, purpose-built tools. +- **Two variables separate winners from losers**, per Thorsten: (1) knowing how to use agents, and (2) **having the token budget to do it.** Scope matters on the second, and was corrected here 2026-07-28: he is describing **metered** fleet work (parallel orbs, parked agents), not a subscription. At individual scale the corpus's own practitioners run large agent setups on flat consumer plans without reporting a ceiling — so for a non-engineer this is a skill divide, and the budget is one subscription. See the scoping note on [[enterprise-ai-reality]]. +- **The velocity claim:** *"You cannot take a programmer who doesn't use AI, they're going to get crushed by a mediocre programmer with AI."* Stated about programmers, but the internal-software argument extends it to any role that has ever maintained a spreadsheet. +- **The hard part is not building.** His printer anecdote is the whole skill in one exchange: asked to build an app so a tablet prints a paper receipt for the kitchen, he answered *"Why do you need a printer? Why not a second tablet?"* Seeing the workflow underneath the request is the surviving competence — see [[product-ownership]] and the first-principles section there. + +**Why this matters for the webinar.** It is the outside evidence for thesis T5 in [[2026-07-28-webinar-theses]] — the claim that had been the talk's least provable. [[eugene]]'s arc (chat box → your own OS) lands on exactly this: "It dissolved — into the operating system. Into little tools you make for yourself… You don't buy it. You build it — one small tool at a time." Thorsten reaches the same endpoint from a frontier-engineering starting point and with a non-technical audience (a 20-person social club, a menu photo, a phone). That convergence is usable evidence: the pitch is not an engineer's fantasy about non-engineers, it is what happens when someone with the skill applies it to an ordinary group of people. + +## Evidence + +- Excel/wiki/hack replacement, the club ordering app in ~2 hours of phone typing, the token-budget variable, "crushed by a mediocre programmer with AI," the printer anti-example — [[2026-07-28-agentic-engineering-10x-developer]]. +- The webinar's convergent framing — "little tools you make for yourself," the OS arc — `raw/notes/Webinar script.md` (raw, not yet ingested). +- Non-engineer capability ceiling and the same build-it-yourself instinct — [[levels-of-ai-usage]], [[personal-ai-operating-system]]. + +## Related Pages + +- Concepts: [[emacsification-of-software]] (sibling mechanism — remixing rather than building), [[maintenance-is-the-real-cost]] (the bounding counterweight), [[personal-ai-operating-system]], [[levels-of-ai-usage]], [[product-ownership]], [[shedding-weight]], [[enterprise-ai-reality]] (token budget as access), [[solve-first-then-skillify]] +- Entities: [[thorsten-ball]], [[eugene]], [[allie-miller]], [[virtido]] + +## Contradictions / Uncertainty + +- **Survivorship.** Thorsten is a founding engineer at an agent company building for a club he belongs to. The corpus's actual non-engineers ([[nina]], [[yulia]], [[larysa]]) hit friction, [[integration-dead-ends|integration dead-ends]] and memory loss well before "2 hours on a phone." His datapoint proves the ceiling is high, not that the floor is low. Status: tentative. +- **Nobody owns the result.** Internal software built in an evening still needs to survive its author leaving, a schema change, or an incorrect order going out. The source treats creation cost as the only cost — the same gap [[emacsification-of-software]] has around maintenance. *(Upgraded 2026-07-29 from self-criticism to a sourced contradiction:* [[2026-07-29-what-if-we-vibe-code-it]] *makes this objection its whole thesis — [[maintenance-is-the-real-cost|the cost of software is maintenance, not writing]] — and supplies the corpus's only observed outcome of this pattern in the wild: a company that built its own Jira clone in March 2026 and returned to a bought tracker by July. Partial reconciliation: the club app passes that source's own build-vs-buy checklist — tiny, no SLA, no external users — so the disagreement is about where the threshold sits, not whether one exists.)* +- **The token-budget variable is named and then dropped.** Who pays, how much, and what happens to people or teams without the budget is unaddressed in the source. Partly resolved by scope (see above): under flat-rate consumer pricing it appears not to bind at individual scale, and the corpus has two practitioners running large setups to show it. It remains open for metered pricing and fleet scale — and [[eugene]]'s prediction that prices *rise* ("what I now buy for 200 will cost about 1,000") would reopen it for everyone if it holds. Status: tentative. + +## Next Questions + +- What is the realistic first internal tool for the webinar's HR audience — and does it survive contact with the friction Nina and Yulia describe? +- Is there a threshold above which internal software must graduate to being owned like a product (an on-call rota, a schema, a backup), and where is it? *(Sharpened 2026-07-29: [[maintenance-is-the-real-cost]] supplies a checklist for the question — size, ongoing support, operational load, second-business willingness — but not the line itself.)* diff --git a/wiki/concepts/harness.md b/wiki/concepts/harness.md index 0a3e7e7..b90e8bd 100644 --- a/wiki/concepts/harness.md +++ b/wiki/concepts/harness.md @@ -16,6 +16,8 @@ The harness is the de-facto unit of agentic work in 2026. A good one has: a **sh **The business-facing formula.** An anonymous Russian business short ([[2026-07-22-ai-is-stupid]]) independently restates the concept for non-engineers: the harness is an "engineering wrapper" — what the model must verify, which tools to trust, how to shape the answer, what is forbidden — and **strong model + your business context + harness = employee-level answer**. Remove any component and you get "smart but generic," "specific but undisciplined," or "stupid AI." Useful as webinar language: it names what the audience already feels (generic answers) without requiring the engineering vocabulary. +**A second reference harness: [[amp]].** Where [[claude-code]] is local-first and user-extended, AMP is sandbox-first and vendor-curated ([[2026-07-28-agentic-engineering-10x-developer]]): a PWA install, a low/medium/high/ultra dial that maps each level to a model *and* a sub-agent set, named sub-agents (**Oracle** the reviewer, **Painter** the image generator), a meta-agent (**Puck**) that spawns and messages other agents, and **orbs** — remote sandboxes where one URL carries thread + agent + computation + diff (see [[async-by-default]]). Two things it demonstrates about the concept: the harness is now the *product* (AMP's most-asked customer question is "what's the meta — what model, what prompt?", i.e. customers pay for research decisions), and a harness can be strong with **no user-authored skills layer at all** — the structure exists, but the vendor supplies it. That is the design axis [[hermes]] and Claude Code put in the user's hands. + **The governance fault line:** [[eugene]] argues every developer should **build their own** harness (deep knowledge → more effective). [[sebastian]] counters that "bring your own harness" cannot survive enterprise compliance — it must be a company-managed resource, and *that gap is the business*. See [[enterprise-ai-reality]]. ## Evidence @@ -26,16 +28,18 @@ The harness is the de-facto unit of agentic work in 2026. A good one has: a **sh - Claude Code as the reference harness across surfaces — [[2026-07-14-gap-between-ai-users-irreversible]]. - Consolidated multi-project workspace, inter-agent messaging, completion signals, "before and after" claim — [[2026-07-21-larysa-interview]]. - Harness as "engineering wrapper"; model + context + harness formula; "stupid AI" as the harness-less default — [[2026-07-22-ai-is-stupid]]. +- AMP's dial/sub-agents/meta-agent/orbs; "what's the meta?" as the customers' recurring question — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages -- Tools: [[claude-code]], [[hermes]] -- Concepts: [[skills-as-memory]], [[agentic-loops]], [[evolution-of-agent-tooling]], [[context-as-scarce-resource]], [[enterprise-ai-reality]], [[personal-ai-operating-system]], [[leave-less-room-for-imagination]] -- Entities: [[eugene]], [[sebastian]], [[konstantin]], [[larysa]] +- Tools: [[claude-code]], [[hermes]], [[amp]] +- Concepts: [[skills-as-memory]], [[agentic-loops]], [[evolution-of-agent-tooling]], [[context-as-scarce-resource]], [[enterprise-ai-reality]], [[personal-ai-operating-system]], [[leave-less-room-for-imagination]], [[async-by-default]], [[shedding-weight]] +- Entities: [[eugene]], [[sebastian]], [[konstantin]], [[larysa]], [[thorsten-ball]] ## Contradictions / Uncertainty -- Personal vs company-managed harness is an unresolved tension (Eugene vs Sebastian), not a settled answer. +- Personal vs company-managed harness is an unresolved tension (Eugene vs Sebastian), not a settled answer. [[amp]] adds a third option neither of them argues for: a **vendor-managed** harness, where the research decisions are the purchase and the user tunes almost nothing. +- **Local vs remote.** Eugene's consolidation pitch assumes one place *on your machine*; Thorsten predicts local dev disappears into remote sandboxes. Compatible only if the thing being consolidated is the interface rather than the compute. Status: tentative. - The "life split into before and after" consolidation payoff is self-reported by its builder and never measured; Larysa, the practitioner it was pitched to, does not yet run one. Status: tentative. ## Next Questions diff --git a/wiki/concepts/leave-less-room-for-imagination.md b/wiki/concepts/leave-less-room-for-imagination.md index 0b4fbdd..baeba43 100644 --- a/wiki/concepts/leave-less-room-for-imagination.md +++ b/wiki/concepts/leave-less-room-for-imagination.md @@ -17,6 +17,8 @@ This is Eugene's explicit critique of demo culture: asking Claude to build a who **Model-choice corollary.** Eugene runs **Claude 4.7** rather than 4.8, calling 4.8 "too proactive" — "without the flights of fancy 4.8 has." He treats over-eagerness as a property to select against in the model, not only in the prompt. (Whether that is really a model trait or an unspecified-prompt symptom is unresolved — see below.) +**A worked example of the constructive form.** [[thorsten-ball]]'s prompt for porting a feature to the CLI ([[2026-07-28-agentic-engineering-10x-developer]]) shows what "less room" looks like without a skill: **set the standard** ("look at how it's implemented in web UI") → **state intent** ("I want to port this to our CLI") → **riff on the design** (name the commands, guess at the modality, invite disagreement) → **specify process** ("research how it's implemented, research how we communicate it, document how it works, sit down and think, compile what you learned, *then* come up with a good idea") → **set constraints and economics** ("Fable is expensive — use GPT models for the implementation, then present the results"). His summary: *"This is how I would talk to a senior engineer. This is the Slack message I'd send."* Note that the *process* step is doing most of the work here — it removes imagination about **how to proceed**, not only about what to build. Worth holding against this page's remedy: Eugene freezes procedure into a [[skills-as-memory|skill]]; Thorsten retypes it, and rejects skills outright (see that page's contradictions). + Tension worth holding: [[think-wider-not-bigger]] argues for giving models *more* latitude across a wider surface. These are compatible only if read as breadth-of-attempts vs. tightness-of-each-spec — many cheap wide attempts, each individually well-constrained. ## Evidence @@ -24,17 +26,19 @@ Tension worth holding: [[think-wider-not-bigger]] argues for giving models *more - "The more room for imagination, the more it will exploit it"; the collateral-damage-you-won't-notice framing; the one-or-two-requests demo critique; 4.7 vs 4.8 — [[2026-07-21-larysa-interview]]. - "Narrow the variability of interpretation when prompting" as a plateau practice — [[2026-07-14-yulia-interview]]. - Skills as frozen, proven procedure — [[2026-07-14-skills-based-on-git]]. +- The five-part prompt structure (standard / intent / riff / process / economics); "the Slack message I'd send to a senior engineer"; stop tuning model choice — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages -- Concepts: [[skills-as-memory]], [[solve-first-then-skillify]], [[levels-of-ai-usage]], [[integration-dead-ends]] (the capability-side mirror), [[think-wider-not-bigger]] (tension), [[product-ownership]] -- Entities: [[eugene]], [[larysa]], [[claude-code]] +- Concepts: [[skills-as-memory]], [[solve-first-then-skillify]], [[levels-of-ai-usage]], [[integration-dead-ends]] (the capability-side mirror), [[think-wider-not-bigger]] (tension), [[product-ownership]], [[context-as-scarce-resource]] +- Entities: [[eugene]], [[larysa]], [[claude-code]], [[thorsten-ball]] ## Contradictions / Uncertainty - Sits in tension with [[think-wider-not-bigger]]; reconciled above as breadth vs. per-task tightness, but neither source addresses the other. Status: tentative. - **Diff summaries vs invisible drift** (added 2026-07-24): Theo/Dax recommend routing big diffs through agent per-file summaries instead of line-by-line reads — "anything weird will stick out" ([[2026-07-24-youre-reading-way-too-much-code]]). Eugene's claim here is the opposite: the damage is what you *don't* notice, and a summary is exactly where drift hides. Theo's tier framework partially reconciles it (summaries are a tier-B/C practice; tier-D still reads every line, and slop verification catches what reading misses — see [[make-more-cheap-code]]), but neither source addresses the other. Status: tentative. - "4.8 is too proactive" is one practitioner's preference from production use, not a benchmark. Status: tentative. +- **Model choice as a lever, or a distraction?** (added 2026-07-28.) Eugene selects *against* over-proactivity at the model level (4.7 over 4.8). [[thorsten-ball]] says the opposite about the whole activity: past a frontier model there are diminishing returns on which one you pick and even on the effort level, and "if you're mad your model doesn't use camelCase, rethink your software engineering, not the model" — put the effort into the information you supply ([[context-as-scarce-resource]]). They are reconcilable if Eugene's complaint is about *behaviour under an underspecified prompt* rather than capability, which is exactly what Thorsten would say the prompt should fix. Neither addresses the other. Status: tentative. ## Next Questions diff --git a/wiki/concepts/levels-of-ai-usage.md b/wiki/concepts/levels-of-ai-usage.md index 50c9a2c..d0d8205 100644 --- a/wiki/concepts/levels-of-ai-usage.md +++ b/wiki/concepts/levels-of-ai-usage.md @@ -33,5 +33,7 @@ Supporting practices at the plateau: keep CLAUDE.md self-maintaining ("always ke ## Next Questions -- Does the final webinar script keep this exact rung order? (`raw/sources/Webinar script.md` is not yet ingested.) +- **Is the top rung the right ceiling?** (Raised 2026-07-28 by lint.) This ladder's non-programmer ceiling is *CLAUDE.md + skills*, and [[thorsten-ball]] reaches the frontier with neither — no skills, no MCP, no slash commands, context in the codebase and `AGENTS.md` ([[2026-07-28-agentic-engineering-10x-developer]]). If his practice generalises, the ladder's top two rungs are a detour rather than a summit; if it doesn't, the reason is that he has a codebase to encode context into and this ladder's audience does not — which would be worth stating *as* the rung's precondition. See [[skills-as-memory]] for the three competing readings. + +- Does the final webinar script keep this exact rung order? (`raw/notes/Webinar script.md` — an authored deliverable, not an ingest candidate.) - Where do agents/processes (the harness's outer loops) sit for a non-programmer — above skills, or out of reach? diff --git a/wiki/concepts/maintenance-is-the-real-cost.md b/wiki/concepts/maintenance-is-the-real-cost.md new file mode 100644 index 0000000..3a255ad --- /dev/null +++ b/wiki/concepts/maintenance-is-the-real-cost.md @@ -0,0 +1,37 @@ +# Maintenance Is the Real Cost + +#concept + +## Summary + +The cost of software was never in writing it — it is in **running it after the first user arrives**. AI collapsed the writing cost, which was always the small part, and left the real cost untouched. The trap of the vibe-coding era: "assemble in two weeks — easy; carry it forward — impossible." An internal service is an internal business. + +## Current Understanding + +- **The misconception being corrected:** "we couldn't build our own Jira before, and now with AI we can." False on both ends — developers always could (by hand, with a team, in months); they didn't because they didn't want to *operate* the result. The blocker was never capability. +- **What arrives with the first user:** bugs and regressions, feature requests, "something's not working / working wrong / didn't work," logs, monitoring, on-call, uptime responsibility. The project built "to save on a subscription" becomes a standing job with dedicated people — exactly the job the vendor was paid to do. +- **The two-business paradox:** a company whose product is X, quietly carrying a self-hosted tracker/logger, is running two IT businesses. Bad for the company (pays for one product, staffs two) and for the developer (a primary job you're blamed for neglecting, plus a secondary one you're blamed for neglecting). +- **The pendulum case:** March 2026 — a company builds its own Jira clone and migrates; July 2026 — the same (or a similar) company returns to a bought tracker (Linear). Status: tentative (second-hand tweets, fuzzy identification), but it is the corpus's only *observed outcome* of the build-your-own-tools thesis, and it's a reversal. +- **The build-vs-buy checklist** (prescriptive): rewrite only small, non-evolving dependencies; if it needs ongoing support, cost it as a separate project; count the operational load; ask whether the business wants a second IT business inside itself. Otherwise keep paying the vendor — the money buys operational offload, not code. +- **Where it agrees with the vault's spine:** "writing code was never the bottleneck" is the same premise as [[harness]]-over-model and [[make-more-cheap-code]]'s verification bottleneck. The corpus now has three candidates for the *real* bottleneck — context (Thorsten), verification (Theo), maintenance (this source) — which are not rivals: they are the costs at authoring time, at shipping time, and over the artifact's lifetime, respectively. + +## Evidence + +- All claims, the pendulum case, the checklist, the Datadog self-report — [[2026-07-29-what-if-we-vibe-code-it]]. +- The objection was already latent in the vault before this source named it: [[explosion-of-internal-software]] ("nobody owns the result") and [[emacsification-of-software]] ("maintenance is assumed away") both flagged it as their own weakest point. + +## Related Pages + +- Concepts: [[explosion-of-internal-software]] (the thesis this bounds), [[emacsification-of-software]] (forks age too), [[code-as-throwaway]] (throwaway is safe *because* unmaintained), [[make-more-cheap-code]] (the ship/no-ship line is also the maintain/no-maintain line), [[shedding-weight]] (the inverse move — deleting owned software rather than acquiring it), [[product-ownership]] (owning an outcome includes owning its ops) +- Entities: [[thorsten-ball]] (the predictions this bounds) + +## Contradictions / Uncertainty + +- **vs. [[explosion-of-internal-software]] / [[thorsten-ball]]:** Thorsten predicts teams remix mid-size software ("Riverside but audio-only") and the Excel layer becomes real tools; this source's pendulum case is that pattern failing in the wild. Partial reconciliation: the club app *passes* this source's own checklist (tiny, no SLA, no external users) — the disagreement is only about where the threshold sits, not whether one exists. Recorded on both pages. +- **Does the objection survive agents doing the maintenance?** The author assumes ops load lands on humans. The corpus's outer-loop material ([[agentic-loops]], [[async-by-default]]) implies agents could absorb some of it — but no source demonstrates agent-carried ops for an internal service, and [[async-by-default]]'s own caveat (proof produced by the thing being checked is not verification) cuts against trusting it blind. Open. +- The pendulum case is one anecdote, second-hand. The author's own company is *currently* building a Datadog replacement — if it ships and survives, he becomes his own counterexample. Status: tentative. + +## Next Questions + +- Where exactly is the graduation threshold — the point at which a personal/internal tool must be owned like a product (on-call, schema, backups)? [[explosion-of-internal-software]] asks the same question; this source supplies the checklist but not the line. +- For the webinar's HR audience: which of their candidate tools (candidate knowledge base, transcribe→summarize) fall on the safe side of the checklist, and which quietly cross into "second business"? diff --git a/wiki/concepts/make-more-cheap-code.md b/wiki/concepts/make-more-cheap-code.md index f4f4dcc..5d10af5 100644 --- a/wiki/concepts/make-more-cheap-code.md +++ b/wiki/concepts/make-more-cheap-code.md @@ -15,16 +15,19 @@ - **Exploration patterns:** slop-port a service to another language just to benchmark it; test 3 theories of an ambiguous PR in parallel; **use dumb-model agents as API usability testers** — if a weak model can't build on your SDK, that's a UX bug in the SDK. - **Reading economics.** Reading still costs attention (the human-side analog of [[context-as-scarce-resource]]): don't read faster, read *only what's worth reading* — every signature and API always, function bodies rarely, per-file agent summaries instead of giant diffs (via Dax). Have AI review code before humans do. - **What this is not:** a license to merge unreviewed slop. Theo explicitly keeps hand-verification of shipped code unchanged and disowns vibe-coders who ship slop ("I hate them too"). +- **Variations, not answers** (added 2026-07-28). [[thorsten-ball]] extends the same economics past code into *design decisions*: ask for 10–15 variants and pick one. His orb icon came from 15 AI-generated versions across styles and 18 palettes; AMP's news imagery from turn-by-turn Midjourney rounds. Generation is cheap, so the human's job moves from producing the artifact to **choosing among artifacts** — taste at AI speed, which is his answer to the slop objection ([[code-as-throwaway]]). The webinar-relevant part: this is the version of "make more cheap code" that needs no codebase, so it transfers directly to a non-engineer ([[2026-07-24-non-engineer-throwaway-verification]]). +- **Ask for proof, since you're waiting anyway.** Screenshots, benchmarks, dark-mode and light-mode variants, fifty tests in parallel — cheap generated artifacts whose only job is to make a claim checkable. See [[async-by-default]], where the practice belongs to delegation rather than to reading. ## Evidence - All claims, ratios, tier table, slop patterns, Dax/Shao citations — [[2026-07-24-youre-reading-way-too-much-code]]. - Groundwork (code disposable, kill without guilt, G-brain markdown tier) — [[2026-07-14-everything-we-knew-about-software-has-changed]]. +- 15 icon variants, Midjourney rounds, "ask the agent for proof" — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages -- Concepts: [[code-as-throwaway]] (parent claim: cost → zero; this page is its *discipline* — what cheap code is actually for), [[think-wider-not-bigger]] (same breadth logic applied to generation volume rather than ambition), [[product-ownership]] (verifying as the human's remaining job), [[solve-first-then-skillify]] (contrast: slop is frozen into nothing; skills freeze the procedure), [[leave-less-room-for-imagination]] (tension — see below), [[context-as-scarce-resource]] -- Entities: [[theo-browne]], [[eugene]] +- Concepts: [[code-as-throwaway]] (parent claim: cost → zero; this page is its *discipline* — what cheap code is actually for), [[think-wider-not-bigger]] (same breadth logic applied to generation volume rather than ambition), [[product-ownership]] (verifying as the human's remaining job), [[solve-first-then-skillify]] (contrast: slop is frozen into nothing; skills freeze the procedure), [[leave-less-room-for-imagination]] (tension — see below), [[context-as-scarce-resource]], [[async-by-default]] (proof artifacts as the delegated form of the same move) +- Entities: [[theo-browne]], [[eugene]], [[thorsten-ball]] ## Contradictions / Uncertainty @@ -35,4 +38,5 @@ ## Next Questions - What does the throwaway-verification bucket look like in a non-engineer's workflow (the webinar audience) — is there an HR/BA analog of "10,000 lines of slop to verify one line"? *(Answered by synthesis 2026-07-24: generated checks, not generated content — fresh-agent misread tests, parallel interpretations, checker skills, synthetic-candidate simulations. See [[2026-07-24-non-engineer-throwaway-verification]].)* -- Does tier-A slop generation stay cheap once context is accounted for — or does reviewing *agent behavior* replace reviewing code as the attention sink? +- Does tier-A slop generation stay cheap once context is accounted for — or does reviewing *agent behavior* replace reviewing code as the attention sink? *(Sharpened 2026-07-28: [[async-by-default]] multiplies parallel agents without multiplying review capacity, and Thorsten names **token budget** as a winner/loser variable — so the honest answer may be that cheap code is cheap in money and expensive in attention, which is precisely the resource this page says is binding.)* +- Does "15 variations, pick one" hold where the choice needs a criterion rather than taste? Picking an icon is judgment you already have; picking among 15 candidate job descriptions or architectures may require the analysis the variations were supposed to replace. diff --git a/wiki/concepts/personal-ai-operating-system.md b/wiki/concepts/personal-ai-operating-system.md index 780214a..8fda2cc 100644 --- a/wiki/concepts/personal-ai-operating-system.md +++ b/wiki/concepts/personal-ai-operating-system.md @@ -20,21 +20,26 @@ Three layers, built bottom-up: Mindset reframes: AI as **first-class teammate** (not intern), as an **OS** (not a tool you open), and **[[context-as-scarce-resource|context engineering]]** (not prompt engineering). The 4-tier ladder of AI work: Microtask → Companion → Delegate → Teammate. This is the non-engineer's counterpart to the [[harness]]. +**The fourth layer the corpus keeps circling: tools you build for yourself.** Allie's three layers are all *context and orchestration*; [[explosion-of-internal-software]] adds the artifact — a real, small piece of software replacing the spreadsheet. [[thorsten-ball]] supplies the non-engineer-shaped proof (a 20-person club's ordering process, ~2 hours of phone typing) and [[eugene]]'s webinar arc lands on the same place: "little tools you make for yourself." The interface question is where they differ — Thorsten deletes UI so he can prompt ([[build-for-the-agent-not-the-human]]); the OS framing builds a tiny UI so you need not prompt. Both agree the *generic* surface, chat box or admin panel, is what disappears. + ## Evidence - 3 foundation docs, 4 surfaces, "just complain," proactive workflows, 4-tier model, trust calibration — [[2026-07-14-gap-between-ai-users-irreversible]]. - Setup scale: 36 workflows, ~28 master agents, ~100 agents; 2–10× productivity. +- Independent convergence on tools-you-build-for-yourself, from a frontier engineer applying it to a non-technical group — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages -- Concepts: [[skills-as-memory]], [[context-as-scarce-resource]], [[connections-as-moat]], [[levels-of-ai-usage]] -- Entity: [[allie-miller]], [[eugene]] -- Tools: [[claude-code]] +- Concepts: [[skills-as-memory]], [[context-as-scarce-resource]], [[connections-as-moat]], [[levels-of-ai-usage]], [[explosion-of-internal-software]], [[build-for-the-agent-not-the-human]], [[async-by-default]] (scheduled workflows are its non-engineer form) +- Entity: [[allie-miller]], [[eugene]], [[thorsten-ball]] +- Tools: [[claude-code]], [[amp]] - Compare: [[harness]] (engineer's version of the same "universal agent + context" idea — see its *consolidation over tool-hopping* section, where Eugene independently arrives at the same "everything in one place" OS framing from the [[2026-07-21-larysa-interview|Larysa interview]]) ## Contradictions / Uncertainty - "Investment not cost" (1 hour → ~3 hrs/week saved) is Allie's framing; the payback is asserted, not independently measured. Status: tentative. +- **Persistent context vs memory-as-anti-feature** (recorded here 2026-07-28 by lint; previously logged only on [[skills-as-memory]]). Allie's layer 1 is persistent context documents, used without complaint. [[eugene]] holds that built-in agent memory is a net negative — "it gives no benefit and confuses users to hell… it'd be better if it didn't [exist]" ([[2026-07-21-larysa-interview]]). The two are largely reconcilable — both prefer *authored* context over *inferred* context, and Allie's docs are authored — but the blanket condemnation is one voice and this page never registers it. Status: tentative. +- **Does the OS need a skills layer at all?** [[thorsten-ball]] runs a 99%-AI-written codebase with no skills, no MCP and no slash commands ([[2026-07-28-agentic-engineering-10x-developer]]), which challenges layer 2 of this stack directly. Three competing readings are logged on [[skills-as-memory]]; the one most favourable to this page is that his context lives in a codebase he owns, and Allie's audience has none. Status: tentative. ## Next Questions diff --git a/wiki/concepts/product-ownership.md b/wiki/concepts/product-ownership.md index be9bbbb..aa10b83 100644 --- a/wiki/concepts/product-ownership.md +++ b/wiki/concepts/product-ownership.md @@ -12,21 +12,26 @@ The durable human skill in the AI era: **owning outcomes**, not completing ticke - **Reframe the vocabulary:** stop thinking in tasks/tickets; think in problems and desired outcomes. If you don't understand what to build, "you will simply not be an engineer anymore"; if you *do* get closer to the product, the software gets better (product-wise, even if not always technically). - **Verification is the new craft.** As [[code-as-throwaway|code becomes disposable]], the engineer's value is *directing and verifying* — Sebastian's printer anecdote: his edge was knowing how to instruct and check the result, not writing Java. This is also the senior's advantage: **read what you approve** ([[seniority-and-the-junior-squeeze]]). - **Allie's parallel:** the meta-skill is **knowing what good looks like** (taste) — you don't need to do the graphic design to judge whether the ad is good. +- **First-principles thinking becomes the top skill** ([[thorsten-ball]]). His anti-example: someone at his club asked for an app so a tablet prints a paper receipt the kitchen picks up. His push-back — *"Why do you need a printer? Why not a second tablet?"* — is the whole competence in one question. The skill is **seeing the workflow underneath the request**, and it is now the scarce half of the job: *"everyone becomes an architect,"* and the value is knowing solutions from other industries and having the right idea for this problem. Note the direction of travel: the profile-picture story above is a failure to check the *output*; the printer story is a failure to check the *problem*. Ownership now runs at both ends. +- **What just got commoditised.** What senior engineers used to hand-teach over 2–3 years — his example, the safe multi-step column-drop migration — is a 30-second model output. Also gone is the lucky overlap of the last 20–30 years, where "the guy who loves Haskell on weekends is also the guy who models the finance backend well." Technical depth and domain judgment have come apart, and only the second is still scarce. See [[seniority-and-the-junior-squeeze]]. ## Evidence - Profile-picture story, "problems not programmers," printer anecdote, ownership as mindset — [[2026-07-14-sebastian-eugene-interview]]. - "Knowing what good looks like" / taste as the meta-skill — [[2026-07-14-gap-between-ai-users-irreversible]]. +- The printer/tablet push-back, "everyone becomes an architect," the 2–3-years-to-30-seconds migration example, the dissolving Haskell/finance-backend overlap — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages -- Concepts: [[code-as-throwaway]], [[seniority-and-the-junior-squeeze]], [[connections-as-moat]], [[decoupling-identity-from-profession]] -- Entities: [[sebastian]], [[eugene]], [[allie-miller]] +- Concepts: [[code-as-throwaway]], [[seniority-and-the-junior-squeeze]], [[connections-as-moat]], [[decoupling-identity-from-profession]], [[explosion-of-internal-software]] (where the framing skill gets exercised), [[shedding-weight]] (deciding what should exist at all), [[make-more-cheap-code]] +- Entities: [[sebastian]], [[eugene]], [[allie-miller]], [[thorsten-ball]] ## Contradictions / Uncertainty - "Get closer to the product" can improve product quality while *reducing* technical quality — the interview flags this trade-off explicitly. +- "Everyone becomes an architect" is asserted, not argued. The corpus's own non-engineers reach for AI to do the framing *for* them ([[levels-of-ai-usage]]), and nothing establishes that first-principles thinking distributes more widely than the technical skill it replaces. Status: tentative. ## Next Questions - How do you *teach/hire for* ownership if it's a personality trait, not a checklist? +- If the senior's 2–3 years of hand-taught pattern knowledge is now a 30-second output, what is the new apprenticeship — and does it produce the judgment the printer story requires? Compounds the pipeline problem on [[seniority-and-the-junior-squeeze]]. diff --git a/wiki/concepts/seniority-and-the-junior-squeeze.md b/wiki/concepts/seniority-and-the-junior-squeeze.md index 37e5100..55f670f 100644 --- a/wiki/concepts/seniority-and-the-junior-squeeze.md +++ b/wiki/concepts/seniority-and-the-junior-squeeze.md @@ -12,20 +12,24 @@ Counter-intuitively, AI has *raised* demand for seniors and made juniors "comple - **The junior risk (a security argument):** the habit of clicking "yes… yes… allow for all future" is how "API keys are leaked, databases get dumped or deleted." A junior can't evaluate a 250-line bash script; a senior at least *could*. "Give a junior fresh out of university access to this almighty Claude and… the codebase — they will [wreck] it in two days." **Read what you approve.** - **Team shape:** the ~8-person scrum team (scrum master + PM + requirements engineer + big dev team) collapses to **2–3 people** — one coordination/ownership role plus one or two who manage the coding agents, sharing responsibilities. - **Leveling caveat:** on *pure programming skill*, AI **levels** senior and junior (same output). The senior's edge is entirely in judgment, verification, and knowing failure modes — not typing speed. Contrast with [[connections-as-moat]], where the edge is relationships. +- **What the levelling actually consumed** ([[thorsten-ball]], 2026-07-28): the transferable content of seniority. "What senior engineers used to hand-teach in 2–3 years" — his example is the safe multi-step column-drop migration — "is now a 30-second model output." So the *knowledge* half of seniority is commoditised while the *judgment* half is not, which is a sharper version of this page's claim and a harsher one for the pipeline: juniors are squeezed out of the entry-level work **and** the apprenticeship that work used to constitute. He also notes the end of a lucky historical overlap: the person who loved Haskell on weekends was also the person who modelled the finance backend well; those two skills have now come apart, and only the domain half is scarce. +- **The floor rose, not just the ceiling.** *"You cannot take a programmer who doesn't use AI, they're going to get crushed by a mediocre programmer with AI."* Read alongside the levelling caveat, the competitive line is no longer senior-vs-junior but tooled-vs-untooled — which is the same claim [[allie-miller]] makes about the irreversible gap, stated about professionals rather than individuals. ## Evidence - Seniors more valuable, juniors squeezed, "allow-all" security habit, junior-wrecks-it-in-2-days, team collapse to 2–3 — [[2026-07-14-sebastian-eugene-interview]]. +- 2–3 years of hand-taught knowledge → 30-second output; the dissolving Haskell/finance overlap; "crushed by a mediocre programmer with AI"; "don't compare yourself to the 1%" — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages -- Concepts: [[product-ownership]], [[enterprise-ai-reality]], [[connections-as-moat]], [[code-as-throwaway]] -- Entities: [[sebastian]], [[eugene]] +- Concepts: [[product-ownership]], [[enterprise-ai-reality]], [[connections-as-moat]], [[code-as-throwaway]], [[explosion-of-internal-software]], [[shedding-weight]] +- Entities: [[sebastian]], [[eugene]], [[thorsten-ball]] ## Contradictions / Uncertainty - Tension: if a junior + Claude can match a senior's output, "juniors are irrelevant" may reflect *today's* hiring psychology more than a permanent truth — and it raises an unspoken pipeline problem (where do future seniors come from?). Status: tentative. +- **Don't generalise from the 1%.** Thorsten's caution cuts across this whole page: online debates cite Mitchell Hashimoto (Ghostty, a GPU-accelerated terminal emulator) as proof AI "isn't good enough," but that is one of the best programmers alive on atypical software. Most software is CRUD, "MySQL and something-something," which agents handle fine. The senior's judgment premium is real where failure is expensive and thinner than the discourse suggests everywhere else. ## Next Questions -- If juniors can't get in, how does the industry produce the next generation of seniors? +- If juniors can't get in, how does the industry produce the next generation of seniors? *(Compounded 2026-07-28 — the apprenticeship content itself is now a 30-second output, so the answer cannot be "they'll learn it on the job.")* diff --git a/wiki/concepts/shedding-weight.md b/wiki/concepts/shedding-weight.md new file mode 100644 index 0000000..dfb69e4 --- /dev/null +++ b/wiki/concepts/shedding-weight.md @@ -0,0 +1,44 @@ +# Shedding Weight + +#concept + +## Summary + +[[thorsten-ball]]'s operating discipline: **most of your process exists because humans used to be the bottleneck, and it should be deleted.** Not optimized — deleted. "Don't optimize for what looks safe today, optimize for the ability to move fast tomorrow." + +## Current Understanding + +The test for any workflow, feature or artifact: *would this exist if agents had always been available?* If not, it is weight. + +Named casualties from the source: + +- **Backlogs.** Old loop: bug reported → backlog → weeks later someone decides it's worth doing → estimate → maybe fix. New loop: *"optimistically spawn these agents, have them parked somewhere, then go through the bug fixes."* You no longer estimate whether a bug is worth fixing when the fix ran while you slept. **Backlogs are an artifact of expensive humans.** +- **CI that re-runs the agent's own tests.** The agent is already in an isolated sandbox and already ran the tests; pushing so CI can repeat them for ten minutes is waste motion. +- **IDE extensions.** [[amp]] killed its VS Code extension — "who has the editor open anymore?" +- **Admin panels and forms.** They existed so that *no code had to change*; changing the code is now cheaper. See [[build-for-the-agent-not-the-human]]. +- **Local dev environments.** Predicted to go away as remote sandboxes take over ([[async-by-default]]). + +The corporate version is aggressive self-cannibalization: AMP publicly kills its own features and accepts churn from users pushed out of their comfort zone. The commercial logic is that a defensible-looking 2025 product ("single agent in a VS Code sidebar with enterprise permissions and per-line attribution") would have been obsolete within a year. + +**The honest form of the exercise** is his suggested internal doc: *"Software Is Dead — Now What?"* — be specific about which of your processes only survive because humans used to be the bottleneck. + +## Evidence + +- Backlogs, CI, VS Code extension, "AMP Frontier Corporation," the frontier bet, the suggested internal doc — [[2026-07-28-agentic-engineering-10x-developer]]. +- The same instinct one layer down: killing code without guilt and resetting rather than guilt-merging — [[2026-07-14-everything-we-knew-about-software-has-changed]], via [[code-as-throwaway]]. + +## Related Pages + +- Concepts: [[code-as-throwaway]] (the artifact-level version of the same move), [[build-for-the-agent-not-the-human]], [[async-by-default]], [[explosion-of-internal-software]], [[product-ownership]] (deciding what should exist is the surviving job), [[enterprise-ai-reality]] (the strongest counterweight) +- Entities: [[thorsten-ball]], [[amp]], [[theo-browne]] + +## Contradictions / Uncertainty + +- **Compliance doesn't shed.** In [[enterprise-ai-reality|regulated enterprises]], CI, audit trails, backlogs and permission systems are frequently *the deliverable* to a regulator, not overhead. Thorsten is describing a startup on the frontier and never scopes the claim; [[sebastian]]'s clients cannot install their own tools, let alone delete their pipeline. Status: tentative. +- "The agent already ran the tests" assumes you trust the agent's report of its own run. Nothing in the source addresses a lying or truncated test run — an obvious place for [[make-more-cheap-code|generated verification]] to re-enter. +- Deleting the backlog and "parking agents" replaces one queue with another; the source does not say who triages the parked fixes or what that costs in attention. + +## Next Questions + +- What is the smallest safe version of this for a non-frontier team — which single pre-agent workflow gives the biggest return when deleted first? +- Does shedding weight have a floor for a *person* rather than a company? The webinar audience's equivalent of "kill your backlog" is unclear. *(Proposed 2026-07-28: the audience-facing form is the question "which of your processes only exist because **you** were the bottleneck?" — carried as thesis T10 in [[2026-07-28-webinar-theses]], where it is suggested as an opener. Untested on a non-engineer audience.)* diff --git a/wiki/concepts/skills-as-memory.md b/wiki/concepts/skills-as-memory.md index 565a27d..7524d3c 100644 --- a/wiki/concepts/skills-as-memory.md +++ b/wiki/concepts/skills-as-memory.md @@ -14,6 +14,8 @@ A classic skill is "**von Neumann without data**" (code, no data). Adding data + The **method** for populating skills is [[solve-first-then-skillify]]: reach the final solution once, then freeze it (Eugene's variant of the heuristic: any correction loop longer than ~3 messages becomes a skill). The HR interviews add a social payoff: a packaged skill is a **handoff/de-risking asset** — a junior "with not even a third of your HR experience" can deliver a decent result, and the expert can take a vacation. +**The dissent: a frontier practitioner who skips skills entirely.** [[thorsten-ball]] — 99% of his company's code is AI-written — reports "**no custom slash commands, no skills, no MCP servers**" ([[2026-07-28-agentic-engineering-10x-developer]]). His substitute is not weaker context but *differently located* context: the codebase itself, a team-maintained `AGENTS.md`, and a long prompt written "like a Slack message to a senior engineer" (set the standard → state intent → riff on design → specify process → set sub-agent economics). He agrees with this page's premise — "where does the agent get its information from?" is the only question he thinks matters — and rejects its mechanism. Recorded as a live contradiction under *Contradictions* below rather than reconciled. + **The negative case: built-in memory as anti-feature.** The [[2026-07-21-larysa-interview|Larysa interview]] supplies the demand-side reason this architecture exists. Her core frustration is that the agent doesn't carry context between sessions — she re-explains, and re-pays in time and tokens. Eugene's answer is not "better memory" but *no* memory: "Memory is the worst thing agents have — it gives no benefit and confuses users to hell. Why even go there? … The memory exists, but the way it's implemented, it'd be better if it didn't." The claim is that an opaque, always-on memory that silently decides what to recall is worse than nothing, because the user can neither inspect nor correct it — whereas a skill is a file you can read, edit, version and delete. Skills are the memory you *author*. ## Evidence @@ -24,19 +26,27 @@ The **method** for populating skills is [[solve-first-then-skillify]]: reach the - Skill as zip-and-hand-over onboarding asset; "create a skill for this" — [[2026-07-14-nina-interview]]. - ~3-message correction-loop heuristic; skills as the non-programmer ceiling (with CLAUDE.md) — [[2026-07-14-yulia-interview]]. - Cross-session memory loss as the #1 practitioner pain; "memory is the worst thing agents have"; skills committed as the webinar remedy — [[2026-07-21-larysa-interview]]. +- Counter-evidence: no skills, no MCP, no slash commands at a 99%-AI-written company; `AGENTS.md` + codebase + rich prompt as the substitute — [[2026-07-28-agentic-engineering-10x-developer]]. ## Related Pages - Concepts: [[evolution-of-agent-tooling]] (tools → MCP → skills), [[harness]], [[context-as-scarce-resource]], [[agentic-loops]], [[personal-ai-operating-system]], [[solve-first-then-skillify]], [[levels-of-ai-usage]], [[leave-less-room-for-imagination]] -- Tools: [[hermes]], [[claude-code]] -- Entities: [[konstantin]], [[allie-miller]], [[eugene]], [[larysa]] +- Tools: [[hermes]], [[claude-code]], [[amp]] (a harness with vendor-curated sub-agents and no user-authored skills layer) +- Entities: [[konstantin]], [[allie-miller]], [[eugene]], [[larysa]], [[thorsten-ball]] ## Contradictions / Uncertainty +- **Skills vs RAG** (recorded here 2026-07-28 by lint; previously logged only on [[context-as-scarce-resource]]). This page's Summary asserts load-on-activation beats RAG pre-injection as settled mechanism, but [[2026-07-22-ai-is-stupid]] names **RAG and long-term assistant memory** as *the* practical context mechanisms for a business audience. Possibly not a real disagreement — business *data* may want RAG where *procedures* want skills — but the corpus has never separated the two cases. Status: tentative. - No standards yet for *what* data to put in a skill or its size limit (Konstantin: 200 GB in one, 100 KB in another, both fine — "ceiling not found"). Status: tentative. - "Built-in memory is a net negative" is Eugene's strong position, not a corpus consensus — [[allie-miller]]'s [[personal-ai-operating-system]] happily uses persistent context docs and never condemns the memory feature. The two are reconcilable (both prefer *authored* context to *inferred* context), but the blanket "better if it didn't exist" is one voice. Status: tentative. +- **Skills may be unnecessary at the frontier** (added 2026-07-28). [[thorsten-ball]] ships 99%-AI-written code with no skills, no MCP and no slash commands. Three readings, none settled by the corpus: + 1. **Situational.** He works daily in *one codebase he controls*, where context can live in the code and `AGENTS.md`. Skills earn their keep when work is spread across many ad-hoc tasks with no codebase to encode into — which is exactly the corpus's HR/BA audience ([[nina]], [[yulia]], [[larysa]]). Under this reading both are right and the disagreement is about who is speaking. + 2. **The abstraction is premature.** Skills are scaffolding for models that needed it; a strong model plus a rich prompt plus a good repo may simply beat a skills library, making the whole layer a 2025 artifact. This is the uncomfortable reading for the webinar's central promise. + 3. **He has skills under another name.** AMP's Oracle/Painter/Puck sub-agents and a maintained `AGENTS.md` *are* curated, reusable, two-stage context — just authored by the vendor and the team rather than the user. Under this reading the dispute is about who curates, not whether curation is needed. + Status: tentative. Note the evidential asymmetry — his is a first-hand report of daily practice at scale, where the pro-skills case rests on Konstantin's architecture argument plus self-reported individual workflows. **Presentation-safe restatement:** [[2026-07-28-webinar-theses]] reframes the claim as *context you author beats context that's inferred*, which holds under all three readings — Konstantin's skills, Allie's foundation docs, Eugene's anti-memory position and Thorsten's `AGENTS.md` are all authored context. ## Next Questions - ~~What's a starter skill set for a non-engineer?~~ Answered in [[2026-07-14-best-first-skill-for-beginner]] (skill-creator as meta-skill; tone-of-voice + anti-AI-language as first content skill). - Do skills actually solve *cross-project* context, or only per-procedure recall? Larysa's complaint may be the former, which skills don't obviously address. +- Is there a test that would separate reading 1 from reading 2 above? The cheapest one available: give a non-engineer the same task with and without a skill and compare drift — the corpus has never run it, and the webinar's promise rests on the answer. diff --git a/wiki/concepts/think-wider-not-bigger.md b/wiki/concepts/think-wider-not-bigger.md index cff5329..100c60a 100644 --- a/wiki/concepts/think-wider-not-bigger.md +++ b/wiki/concepts/think-wider-not-bigger.md @@ -25,6 +25,7 @@ Theo Browne's core reframe: you can't out-improve the models by "getting better" ## Contradictions / Uncertainty - "Bolt a database platform in a day or two" is an ambition claim; reliability parity with incumbents (RDS) is explicitly *not* promised. Status: tentative. +- **Wide latitude vs tight specs** (recorded here 2026-07-28 by lint; previously logged only on the other side). [[eugene]]'s [[leave-less-room-for-imagination]] argues the opposite reflex: every gap you leave gets filled invisibly, so specs should be tightened and frozen into skills. The reconciliation offered there is *breadth of attempts vs tightness of each spec* — many cheap wide attempts, each individually well-constrained — which preserves both, but neither source addresses the other and the reconciliation is the wiki's synthesis, not either author's. Status: tentative. ## Next Questions diff --git a/wiki/entities/amp.md b/wiki/entities/amp.md new file mode 100644 index 0000000..4bf4dc4 --- /dev/null +++ b/wiki/entities/amp.md @@ -0,0 +1,44 @@ +# AMP (Sourcegraph) + +#entity + +## Summary + +Agent product from Sourcegraph, where [[thorsten-ball]] is a founding engineer. The vault's **second reference [[harness]]** after [[claude-code]], and the one that pushes hardest on remote execution: its unit of work is the **orb**, a sleeping remote sandbox tied to one conversation. Officially self-described as "AMP Frontier Corporation." + +## Current Understanding + +**Operating principle:** don't optimize for what looks safe today, optimize for the ability to move fast tomorrow. They could have made money in 2025 building "a single agent in a VS Code sidebar with an enterprise permission system and per-line attribution" — and that model would have been obsolete within a year. Instead they publicly kill their own features on `ampcode.com/news` (the VS Code extension went first). Some users churn; the ones who stay respect the pushing. See [[shedding-weight]]. + +**Shape of the product:** + +- Installed as a **PWA** from `ampcode.com` (Thorsten suspects the acronym itself blocks mass adoption of the install flow). +- A **low / medium / high / ultra dial**, each level mapping to a model *and* a sub-agent choice. Default is medium. +- Sub-agents: **Oracle** (reviewer, gives advice) and **Painter** (generates images). Meta-agent **Puck** controls other agents, spawns orbs, messages them, and runs flows. +- Multi-model: GPT, Anthropic and GLM models all supported. +- **Orbs**: one URL packages the thread + the agent + the computation + the diff; the sandbox sleeps when idle and wakes on typing; the same conversation streams to phone, laptop and TUI; share the URL and a teammate takes over. Agent-to-agent communication shipped recently. See [[async-by-default]]. + +**What customers actually buy:** the most common question AMP gets is "guys, what's the meta? What model, what prompt?" — customers are paying for the research decisions as much as the software. That is a commercial restatement of [[harness|harness-over-model]]. + +**Internal practice:** ~99% of AMP's own code is AI-written; a bug is screenshotted, sent to AMP, and returned as a fix from an orb for spot-check and merge; side-bugs get their own parallel orb, branch and checkout. + +## Evidence + +- Frontier bet, feature-killing, PWA install, model dial, Oracle/Painter/Puck, orbs, multiplayer, velocity anecdotes, live production ship — [[2026-07-28-agentic-engineering-10x-developer]]. + +## Related Pages + +- Entities: [[thorsten-ball]] +- Concepts: [[harness]], [[async-by-default]], [[shedding-weight]], [[build-for-the-agent-not-the-human]], [[code-as-throwaway]] +- Related tools: [[claude-code]] (the vault's other reference harness; local-first where AMP is sandbox-first), [[hermes]] (skills-first, the axis AMP ignores) + +## Contradictions / Uncertainty + +- AMP ships sub-agents and a meta-agent but no user-authored skills layer — a harness design that assumes the *vendor* curates structure, where [[hermes]] and [[claude-code]] assume the *user* does. Whether that is a philosophy or a roadmap gap is unstated. +- "15+ sandbox providers racing to zero margin" is Thorsten's own prediction about the infrastructure his product depends on; he calls it unsustainable without saying what AMP does about it. +- All internal metrics are self-reported by a founding engineer. Status: tentative. + +## Next Questions + +- Does the orb model survive [[enterprise-ai-reality|enterprise compliance]] — code and conversation living in a vendor's remote sandbox is exactly what Sebastian's clients lock down? +- Is "bring your own agent" ([[build-for-the-agent-not-the-human]]) compatible with selling an agent? diff --git a/wiki/entities/claude-code.md b/wiki/entities/claude-code.md index 49298d8..cfa27a7 100644 --- a/wiki/entities/claude-code.md +++ b/wiki/entities/claude-code.md @@ -4,7 +4,7 @@ ## Summary -Anthropic's agentic coding CLI/harness, cited across all four ingested sources as the reference [[harness]]. Used interactively and in fire-and-forget / scheduled modes. +Anthropic's agentic coding CLI/harness, cited across most of the corpus as the reference [[harness]] and used by every practitioner in it except [[thorsten-ball]] (who runs [[amp]]). Used interactively and in fire-and-forget / scheduled modes. ## Current Understanding @@ -29,7 +29,7 @@ Claude Code recurs as the concrete example of the "universal agent" pattern: a s - Concepts: [[harness]], [[skills-as-memory]], [[agentic-loops]], [[context-as-scarce-resource]], [[integration-dead-ends]], [[leave-less-room-for-imagination]] - Entities: [[eugene]], [[larysa]] -- Related tools: [[hermes]] (skills-first harness), Codex CLI, OpenClaude (a Claude Code fork), Cursor, Conductor +- Related tools: [[hermes]] (skills-first harness), [[amp]] (sandbox-first, vendor-curated — the contrasting harness design: orbs instead of a local working directory, Oracle/Painter/Puck instead of user-authored skills), Codex CLI, OpenClaude (a Claude Code fork), Cursor, Conductor ## Contradictions / Uncertainty diff --git a/wiki/entities/theo-browne.md b/wiki/entities/theo-browne.md index eb7c162..f4f3508 100644 --- a/wiki/entities/theo-browne.md +++ b/wiki/entities/theo-browne.md @@ -23,7 +23,7 @@ His second source in the vault sharpens the disposable-code stance into a discip - Concepts: [[think-wider-not-bigger]], [[code-as-throwaway]], [[make-more-cheap-code]], [[decoupling-identity-from-profession]] - Timeline: [[ai-agent-evolution]] -- Compare: [[konstantin]] (model-builder's orchestration view), [[allie-miller]] (personal-OS view) +- Compare: [[konstantin]] (model-builder's orchestration view), [[allie-miller]] (personal-OS view), [[thorsten-ball]] (nearest ally — independently reaches "slop is a human problem" and code-is-cheap from inside a shipping company; where Theo defends cheap code with *verification discipline*, Thorsten defends it with *taste*) - Comparison: [[theo-konstantin-allie]] — three-lens side-by-side (Theo/Konstantin/Allie) ## Contradictions / Uncertainty diff --git a/wiki/entities/thorsten-ball.md b/wiki/entities/thorsten-ball.md new file mode 100644 index 0000000..43eba5b --- /dev/null +++ b/wiki/entities/thorsten-ball.md @@ -0,0 +1,43 @@ +# Thorsten Ball + +#entity + +## Summary + +Founding engineer at [[amp]] (Sourcegraph's agent product); author of *Writing an Interpreter in Go* and *Writing a Compiler in Go*. The corpus's most extreme practitioner datapoint: **99% of the code at his company is written by AI**, and he uses **no skills, no MCP servers, and no custom slash commands**. + +## Current Understanding + +Thorsten's position is that the frontier has moved past tooling questions. What matters is (1) where the agent's information comes from, (2) how agent-friendly your codebase and workflow are, and (3) knowing what you want to build. Everything else — model choice, effort levels, prompt tricks — he treats as diminishing returns or hobby. + +His operating discipline is **[[shedding-weight]]**: aggressively deleting processes and products that only made sense when humans were the bottleneck. AMP publicly kills its own features (the VS Code extension: "who has the editor open anymore?"), and he applies the same knife to backlogs, CI that repeats the agent's tests, and admin panels. + +He is also the corpus's clearest voice on **taste surviving automation**: 99% AI-written code plus 15 AI-generated icon variants he picks between is not slop, because "slop comes from humans not having good product." This is [[theo-browne]]'s position stated from inside a company rather than from a podium — the two are the corpus's closest allies. + +**Where he cuts against the vault:** his answer to "how does the agent know things?" is a good prompt plus a good codebase plus `AGENTS.md` — not [[skills-as-memory|skills]]. He talks to the model "like a Slack message to a senior engineer": set the standard, state intent, riff on design, specify process, set sub-agent economics. Worth holding as a live alternative rather than dismissing: he is a professional engineer working daily in one codebase he controls, which is the situation where codebase-as-context works best and portable skills matter least. + +## Evidence + +- Interview "Agentic Engineering, explained by a 10x developer" (42:33, with David Andre): shed weight, orbs, Emacsification, internal software, first-principles thinking, prompt structure, predictions — [[2026-07-28-agentic-engineering-10x-developer]]. +- Built a club food-ordering app from a menu photo in 3 × 5-minute iterations; encoded a 20-person club's ordering process in ~2 hours of phone typing. +- Forked the diff viewer **hunk** and had AMP add a Gruvbox theme + sidebar file-checkoff in two minutes of agent time; never upstreamed. +- Shipped a change to production live during the podcast recording. + +## Related Pages + +- Tools/orgs: [[amp]] +- Concepts: [[shedding-weight]], [[build-for-the-agent-not-the-human]], [[emacsification-of-software]], [[explosion-of-internal-software]], [[async-by-default]], [[code-as-throwaway]], [[context-as-scarce-resource]], [[product-ownership]] +- Compare: [[theo-browne]] (nearest ally — same slop-is-human, same code-is-cheap stance), [[konstantin]] and [[allie-miller]] (both build the skills layer he skips), [[sebastian]] (the enterprise reality that resists "kill your process") + +## Contradictions / Uncertainty + +- **The skills rejection.** "No custom slash commands, no skills, no MCP servers" is a direct counter to the vault's machine-side spine ([[skills-as-memory]], [[evolution-of-agent-tooling]]). Unresolved; likely scoped to his situation, but he does not scope it himself. Status: tentative. +- **Model choice doesn't matter.** He says stop tuning between models; [[eugene]] deliberately runs 4.7 over 4.8 for over-proactivity ([[leave-less-room-for-imagination]]). Status: tentative. +- The 99%-AI-written figure and the team poll are self-reported from inside a company that sells an agent. Status: tentative. +- "Local dev is going away" is a prediction from someone selling remote sandboxes. +- **The remix/internal-software predictions now have a sourced counter.** [[2026-07-29-what-if-we-vibe-code-it]] argues the cost of software is [[maintenance-is-the-real-cost|maintenance, not writing]], and offers an observed reversal (in-house Jira clone → back to Linear in four months). His club app passes that source's build-vs-buy checklist; his "teams will remix Riverside" prediction is exactly what the checklist rejects. Status: tentative on both sides — one anecdote against one prediction. + +## Next Questions + +- Would he still skip skills if he worked across many unrelated codebases, or as a non-engineer with no codebase to encode context into? +- What is actually in AMP's `AGENTS.md`, and how is it maintained? That file carries the entire load his setup places on it. diff --git a/wiki/lint-reports/2026-07-28-lint.md b/wiki/lint-reports/2026-07-28-lint.md new file mode 100644 index 0000000..17e9413 --- /dev/null +++ b/wiki/lint-reports/2026-07-28-lint.md @@ -0,0 +1,125 @@ +# Lint Report — 2026-07-28 + +#lint-report + +First lint pass on this vault. Scope: all 41 pages in `wiki/` plus `index.md`, checked against the operational contract in `CLAUDE.md`. + +**Vault size at time of check:** 10 sources · 24 concepts · 14 entities · 4 queries · 1 comparison · 1 timeline · 1 overview. + +**Headline:** structurally the vault is in good shape — zero broken links, zero orphans, zero template violations, zero tag errors. The real defects are all *staleness of synthesis*: five references to files that moved, four contradictions recorded on only one of the two pages they concern, and two synthesis pages that predate the last two ingests. Twelve findings; **nine fixed in this pass**, three left as recommendations because they are scope decisions rather than defects. + +--- + +## 1. Clean — verified, no action + +| Check | Result | +|---|---| +| Page-type tags present, correct for folder, on line 3 | **41/41 pass** | +| H1 heading on line 1 | **41/41 pass** | +| Source template sections (6 required) | **10/10 complete** | +| Entity/concept template sections (6 required) | **38/38 complete** | +| Evidence section cites ≥1 `wiki/sources/*` page | **38/38 pass** | +| Broken `[[wiki links]]` | **0** | +| Pages with zero outbound links | **0** | +| Strict orphans (zero inbound) | **0** | +| `index.md` concept list ↔ `wiki/concepts/` files | **24/24 in sync, no drift** | +| `raw/` paths referenced from wiki resolve | 5 failures — see L1 | + +Link-graph health is good. Most-referenced pages: [[eugene]] (33 inbound), [[skills-as-memory]] (29), [[context-as-scarce-resource]] (27), [[claude-code]] (24), [[code-as-throwaway]] and [[harness]] (22 each). Source fan-out median ≈ 17 citing pages. + +--- + +## 2. Fixed in this pass + +### L1 — Stale `raw/` paths after the notes move · **fixed** +The four webinar deliverable docs moved from `raw/sources/` to `raw/notes/` (commit `1361dd7`), but three wiki pages still pointed at the old location. Five references, all now corrected and reworded to say *authored deliverable*, not *not yet ingested* — the distinction the move was making. +- `wiki/concepts/levels-of-ai-usage.md` → `raw/notes/Webinar script.md` +- `wiki/queries/2026-07-22-webinar-theses.md` → three paths +- `wiki/sources/2026-07-14-sebastian-eugene-interview.md` → `raw/notes/Ideas for webinar.md` + +### L2 — Stale source count on [[claude-code]] · **fixed** +Summary read "cited across all four ingested sources" — written when the vault had four. Now rewritten to state the actual situation, which is more useful than a number that will rot again: cited across most of the corpus, and used by every practitioner in it **except** [[thorsten-ball]], who runs [[amp]]. + +### L3 — Dead reference in [[overview]] open questions · **fixed** +The question "should Ideas for webinar / **HR Contacts** / Webinar Plan / Webinar script be ingested?" named a file that does not exist in the vault. The 2026-07-21 ingest caught this and corrected `index.md`, but the fix never propagated to `overview.md`. The question is also now answered — those docs are authored deliverables in `raw/notes/`. Marked resolved, with the phantom file noted. + +### L4–L7 — Contradictions recorded on only one side · **all four fixed** +Rule 5 requires contradictions be recorded explicitly. Four were logged on one page but not on the page holding the opposing view, so a reader arriving from the other direction would see an uncontested claim. + +| # | Contradiction | Was recorded on | Was missing from | +|---|---|---|---| +| L4 | Wide latitude vs tight specs | [[leave-less-room-for-imagination]] (×2) + overview | [[think-wider-not-bigger]] | +| L5 | The skills dissent | [[skills-as-memory]], [[evolution-of-agent-tooling]], [[thorsten-ball]], overview | [[levels-of-ai-usage]] — whose *top rung* is the thing contested | +| L6 | Memory as anti-feature | [[skills-as-memory]], [[allie-miller]] | [[personal-ai-operating-system]] — whose *layer 1* is the thing contested | +| L7 | Skills vs RAG | [[context-as-scarce-resource]] | [[skills-as-memory]] — which asserts the winner in its Summary as settled | + +L5 and L6 are the consequential ones: in both cases the page that *makes* the contested claim was the page not carrying the objection. + +### L8 — [[2026-07-22-webinar-theses]] is two ingests out of date · **staleness note added** +Synthesized from 7 sources; 10 now exist. Preserved rather than rewritten (Update Policy: no silent large rewrites), with a note recording two specific gaps: +1. **Thesis 3 — "Skills are the new memory" — is load-bearing for the talk and now has a live counter-example absent from its own "Honest tensions" list.** +2. Nothing from the last two ingests appears as a candidate: [[make-more-cheap-code]], [[shedding-weight]], and especially [[explosion-of-internal-software]], which is the corpus's strongest *outside* validation of thesis 5 ("you build it, one small tool at a time"). + +### L9 — [[theo-konstantin-allie]] predates Theo's second source and Thorsten · **staleness note added** +Two specific problems, both noted on the page: its reading of Theo's position on code omits the verification discipline he later supplied; and its closing line — "None of the three directly contradicts another; the disagreements in this corpus are elsewhere" — now misleads, because the skills convergence it calls "the vault's strongest cross-source thread" is exactly what the newest source contradicts. + +--- + +## 3. Open findings — recommendations, not defects + +### L10 — "Taste" is the vault's largest uncovered concept · **recommend a page** +Appears in **12 wiki pages / 17 occurrences**, including three entity pages and the vault's one-sentence through-line ("judgment, ownership, **taste**, and in-person relationships"). It is named as *the meta-skill* by [[allie-miller]] ("knowing what good looks like — you don't need to do the graphic design to judge whether the ad is good") and is [[thorsten-ball]]'s entire answer to the slop objection ("taste at AI speed" — 15 generated icon variants, one chosen by hand; "slop = lack of ideas, lack of playfulness"). [[theo-browne]] supplies the third leg via the ship/no-ship line. + +It currently has no page, split across [[product-ownership]] and [[make-more-cheap-code]]. Given that it is one of four nouns in the through-line and the other three all have pages, this is the clearest coverage gap in the vault. + +### L11 — Query pages are a weakly-linked class · **recommend backlinks** +Durable Q&A outputs are not cited back from the concept pages they inform: + +| Query page | Inbound (excl. index/overview) | +|---|---| +| [[2026-07-22-webinar-theses]] | **0** | +| [[2026-07-14-best-first-skill-for-beginner]] | 1 | +| [[2026-07-14-network-from-standing-start]] | 2 | +| [[2026-07-24-non-engineer-throwaway-verification]] | 2 | + +The theses page is effectively orphaned — reachable only from the catalog — despite being the most directly webinar-relevant page in the vault. Rule 7 asks for at least one inbound link; it has none from any content page. + +### L12 — [[2026-07-22-ai-is-stupid]] is thinly integrated · **watch** +Cited by 2 pages against a source median of ~17. Consistent with the 2026-07-22 ingest note that it *restates* the machine-side spine rather than adding to it — so this may be correct rather than a defect. Worth revisiting only if it stays at 2 after the next ingest. + +### L13 — Possible future merge · **watch** +[[emacsification-of-software]] and [[explosion-of-internal-software]] were created in the same ingest, from the same source, as "sibling mechanisms" of one shift (software becomes personal — by remixing vs by building). Both currently carry distinct evidence and distinct open questions. If neither gains independent support from a second source by the next lint, they should merge. + +Also considered and **rejected** as a new page: *trust calibration / what to verify*. It touches 15 pages, but is substantively covered by [[make-more-cheap-code]] (tier discipline), [[code-as-throwaway]] (the auth/payments carve-out) and [[seniority-and-the-junior-squeeze]] (the "read what you approve" habit). Creating it would duplicate rather than consolidate. + +--- + +## 4. Contradiction inventory + +The vault's live disagreements after this pass, all now recorded on both sides: + +1. **Skills as memory vs no skills at all** — Konstantin/Allie/Eugene vs [[thorsten-ball]]. Three competing readings on [[skills-as-memory]]; unresolved, and the evidential asymmetry favours the dissent. The most consequential open question in the vault, because the webinar's central promise rests on it. +2. **Personal vs company-managed vs vendor-managed harness** — [[eugene]] vs [[sebastian]] vs [[amp]]. +3. **Built-in memory as anti-feature** — Eugene vs Allie's untroubled persistent context docs. +4. **Tight specs vs wide latitude** — [[leave-less-room-for-imagination]] vs [[think-wider-not-bigger]]. +5. **Diff summaries as sufficient review vs invisible drift** — Theo/Dax vs Eugene. +6. **Model choice as a real lever vs a distraction** — Eugene (4.7 over 4.8) vs Thorsten (stop tuning). +7. **Local consolidated workspace vs local dev disappearing** — Eugene vs Thorsten. +8. **Skills vs RAG as the context mechanism** — Konstantin vs the business-facing short. +9. **Online vs in-person networking** — Eugene vs Sebastian. + +Items 6 and 7 arrived with the newest source and were recorded at ingest; 1, 3, 4 and 8 were the ones needing reciprocal entries in this pass. + +--- + +## 5. Recommended next actions + +1. **Refresh [[2026-07-22-webinar-theses]]** against all 10 sources — highest value, since it is the page closest to the actual deliverable and it is both stale and orphaned. Fixes L8 and most of L11 in one operation. +2. **Create a `taste` concept page** (L10), consolidating Allie's meta-skill, Thorsten's taste-at-AI-speed, and Theo's ship/no-ship discipline. +3. **Extend or supersede [[theo-konstantin-allie]]** with Thorsten as a fourth lens or explicit dissent column (L9). +4. Consider the falsification test already flagged on [[skills-as-memory]]: same task, with and without a skill, in a non-engineer's hands. It is the cheapest experiment that would move contradiction #1, and nobody in the corpus has run it. + +## Related Pages + +- [[overview]] · [[index]] +- Pages amended by this lint: [[claude-code]], [[levels-of-ai-usage]], [[think-wider-not-bigger]], [[personal-ai-operating-system]], [[skills-as-memory]], [[2026-07-22-webinar-theses]], [[theo-konstantin-allie]], [[2026-07-14-sebastian-eugene-interview]] diff --git a/wiki/overview.md b/wiki/overview.md index a069ea9..e5c817f 100644 --- a/wiki/overview.md +++ b/wiki/overview.md @@ -10,7 +10,7 @@ A high-signal personal knowledge base. `raw/` holds immutable source materials; ## The through-line -Across nine sources — three talks/videos (two of them Theo's), five interviews, and a business-facing short — one spine recurs: +Across eleven sources — five talks/videos/interviews from practitioners (two of them Theo's), five interviews conducted for this project, and a business-facing short — one spine recurs: > **As the cost of writing code goes to zero, value migrates from *producing* software to *directing and verifying* it — and the durable human assets become judgment, ownership, taste, and in-person relationships.** @@ -19,28 +19,33 @@ Everything else hangs off that: - **The machine side** — how the work gets done now: the [[harness]] (universal agent + small toolset + loop), [[skills-as-memory|skills as the new memory]], the tooling progression [[evolution-of-agent-tooling|tools → MCP → skills]], and [[agentic-loops|inner/outer/meta loops]] — all governed by [[context-as-scarce-resource|context as the scarce resource]]. The non-engineer's version is Allie's [[personal-ai-operating-system]]. - **The human side** — what stays yours: [[product-ownership]] over outcomes, [[connections-as-moat|in-person connections]] as the last non-commoditized asset, [[seniority-and-the-junior-squeeze|judgment as risk-reduction]], and the need to [[decoupling-identity-from-profession|decouple identity from profession]]. - **The strategy side** — where to point it: [[think-wider-not-bigger|think wider not bigger]], treat [[code-as-throwaway|code as throwaway]], and mind [[enterprise-ai-reality|enterprise compliance reality]] (the company-managed-harness market). Theo's second video supplies the *verifying* half of the spine its method: [[make-more-cheap-code]] — keep hand-verification of what ships, and generate orders of magnitude more never-shipped code to verify and explore. +- **The frontier side** — what it looks like at the far end, from [[thorsten-ball]] at [[amp]] (99% of their code AI-written): [[shedding-weight|shed weight]] by deleting every process that only existed because humans were the bottleneck; [[build-for-the-agent-not-the-human|build for the agent, not the human]]; work [[async-by-default|async by default]] in remote sandboxes and ask for proof rather than claims. His two mechanisms for software becoming *personal* — [[emacsification-of-software|remixing what exists]] and [[explosion-of-internal-software|building what never did]] — are the corpus's strongest outside validation of the webinar's own thesis, "little tools you make for yourself." He is also its sharpest dissenter: he uses **no skills, no MCP, no slash commands**. Both mechanisms now carry a sourced counterweight — [[maintenance-is-the-real-cost]]: writing code was never the bottleneck, maintenance is, and an internal service is a second business. The reconciliation is a threshold, not a winner: tiny personal tools pass, replacing your Jira does not. - **The demand side** — three interviews ground it all in a real audience. The two HR ones ([[2026-07-14-nina-interview|Nina]], [[2026-07-14-yulia-interview|Yulia]]) supply pain points (interview write-ups, job descriptions, sourcing) that collapse into "a candidate knowledge base plus search," teachable via [[levels-of-ai-usage]] and [[solve-first-then-skillify]]. Their key finding: **adoption is blocked by friction, not resistance.** The [[2026-07-21-larysa-interview|Larysa interview]] adds the *advanced* user's version of the same story: past the friction, the remaining walls are structural — no durable memory, [[integration-dead-ends|integrations that dead-end]], and drift on loose specs ([[leave-less-room-for-imagination]]). Her diagnosis matters because she is technically deep yet skipped the skills rung, which is exactly what her "the agent forgot" complaint reduces to. See [[ai-agent-evolution]] for how the capability curve got here. ## Where sources agree vs diverge -- **Agree:** code is cheap/disposable; harnesses are the unit of work; skills-as-memory (Konstantin ↔ Allie ↔ Eugene); human relationships rise in value (Sebastian ↔ Allie ↔ Eugene, who lands there independently in the Yulia interview); solve-first-then-skillify (Eugene ↔ Konstantin's heuristics); context is the constraint. The [[2026-07-22-ai-is-stupid|"AI is stupid!" short]] independently compresses the machine-side spine into a business one-liner: **model + context + harness = employee-level answer**. -- **Diverge:** personal vs company-managed harness ([[eugene]] vs [[sebastian]]); online vs in-person networking (same pair); OSS as marketing vs OSS growth; built-in agent memory as anti-feature (Eugene) vs persistent context docs used without complaint (Allie); tight specs ([[leave-less-room-for-imagination]]) vs wide latitude ([[think-wider-not-bigger]]); agent diff-summaries as sufficient review (Theo/Dax) vs invisible drift as the core danger (Eugene). These live under "Contradictions" on the relevant pages. +- **Agree:** code is cheap/disposable; harnesses are the unit of work; skills-as-memory (Konstantin ↔ Allie ↔ Eugene); human relationships rise in value (Sebastian ↔ Allie ↔ Eugene, who lands there independently in the Yulia interview); solve-first-then-skillify (Eugene ↔ Konstantin's heuristics); context is the constraint — Thorsten's version is the bluntest: **the dominant variable in output quality is the information you put in**, not the model or the effort level. The [[2026-07-22-ai-is-stupid|"AI is stupid!" short]] independently compresses the machine-side spine into a business one-liner: **model + context + harness = employee-level answer**. Slop is a human problem, not an AI defect (Theo ↔ Thorsten, from verification discipline and from taste respectively). Software becomes personal — "little tools you make for yourself" (Eugene's webinar arc ↔ Thorsten's club app and bespoke forks ↔ Allie's personal OS). +- **Diverge:** personal vs company-managed vs vendor-managed harness ([[eugene]] vs [[sebastian]] vs [[amp]]); online vs in-person networking (Eugene/Sebastian); OSS as marketing vs OSS growth; built-in agent memory as anti-feature (Eugene) vs persistent context docs used without complaint (Allie); tight specs ([[leave-less-room-for-imagination]]) vs wide latitude ([[think-wider-not-bigger]]); agent diff-summaries as sufficient review (Theo/Dax) vs invisible drift as the core danger (Eugene); model choice as a real lever (Eugene runs 4.7 over 4.8) vs a distraction past the frontier (Thorsten); local consolidated workspace (Eugene) vs local dev disappearing into remote sandboxes (Thorsten); build-your-own-tools ([[thorsten-ball]], the webinar arc) vs [[maintenance-is-the-real-cost|buy anything that needs ongoing support]] (the vibe-coding video, with the corpus's only observed reversal: an in-house Jira clone abandoned for Linear in four months). These live under "Contradictions" on the relevant pages. +- **The one that matters most for the webinar:** [[thorsten-ball]] runs a 99%-AI-written codebase with **no skills, no MCP servers and no slash commands** — his context lives in the codebase and `AGENTS.md`. That is the corpus's first credible rejection of the mechanism the webinar's central promise rests on. Three readings (situational / premature abstraction / same thing under another name) are logged on [[skills-as-memory]]; none is settled, and the evidential asymmetry favours him — his is first-hand daily practice at scale. ## Navigation - **[[index]]** — content catalog -- **Sources (9):** [[2026-07-14-everything-we-knew-about-software-has-changed|Theo Browne]] · [[2026-07-14-gap-between-ai-users-irreversible|Allie Miller]] · [[2026-07-14-sebastian-eugene-interview|Sebastian interview]] · [[2026-07-14-skills-based-on-git|Konstantin (git skills)]] · [[2026-07-14-nina-interview|Nina interview]] · [[2026-07-14-yulia-interview|Yulia interview]] · [[2026-07-21-larysa-interview|Larysa interview]] · [[2026-07-22-ai-is-stupid|"AI is stupid!" short]] · [[2026-07-24-youre-reading-way-too-much-code|Theo Browne (reading code)]] -- **People:** [[theo-browne]] · [[allie-miller]] · [[sebastian]] · [[eugene]] · [[konstantin]] · [[nina]] · [[yulia]] · [[larysa]] -- **Tools/orgs:** [[claude-code]] · [[hermes]] · [[virtido]] · [[inspectron]] -- **Concepts:** see the through-line above (18 pages) · **Timeline:** [[ai-agent-evolution]] +- **Sources (11):** [[2026-07-14-everything-we-knew-about-software-has-changed|Theo Browne]] · [[2026-07-14-gap-between-ai-users-irreversible|Allie Miller]] · [[2026-07-14-sebastian-eugene-interview|Sebastian interview]] · [[2026-07-14-skills-based-on-git|Konstantin (git skills)]] · [[2026-07-14-nina-interview|Nina interview]] · [[2026-07-14-yulia-interview|Yulia interview]] · [[2026-07-21-larysa-interview|Larysa interview]] · [[2026-07-22-ai-is-stupid|"AI is stupid!" short]] · [[2026-07-24-youre-reading-way-too-much-code|Theo Browne (reading code)]] · [[2026-07-28-agentic-engineering-10x-developer|Thorsten Ball (agentic engineering)]] · [[2026-07-29-what-if-we-vibe-code-it|"What if we vibe-code it?" (maintenance trap)]] +- **People:** [[theo-browne]] · [[allie-miller]] · [[sebastian]] · [[eugene]] · [[konstantin]] · [[nina]] · [[yulia]] · [[larysa]] · [[thorsten-ball]] +- **Tools/orgs:** [[claude-code]] · [[amp]] · [[hermes]] · [[virtido]] · [[inspectron]] +- **Concepts:** see the through-line above (25 pages) · **Timeline:** [[ai-agent-evolution]] · **Comparison:** [[theo-konstantin-allie]] ## Open Questions (vault-level) - How does an individual build a professional network from a standing start? (Cross-source; the emotional center of the Sebastian interview.) — Tentative protocol drafted at [[network-from-a-standing-start]]; validation instrument at [[2026-07-14-network-from-standing-start]]. - Reusable templates for Allie's 3 foundation docs — a concrete webinar deliverable? -- Should "Ideas for webinar", "HR Contacts", "Webinar Plan" and "Webinar script" be ingested next to connect the corpus to the actual webinar deliverable? (Currently raw-only, per user's ingest scope.) +- ~~Should the webinar docs be ingested to connect the corpus to the deliverable?~~ **Resolved 2026-07-28:** they live in `raw/notes/` as authored deliverables, not sources, and are cited as raw where used. (`HR Contacts.md`, named in the original question, does not exist in the vault.) - Can the transcribe→summarize tool integrate with Manatal (the HR team's ATS)? And is a paid HR-system build going ahead? (Both open from the HR interviews.) - How should a user pre-empt [[integration-dead-ends|integrations that aren't available for their account]]? Both participants in the Larysa interview left this explicitly unsolved — the corpus's only wholly unanswered *technical* problem. - Does the skills rung actually fix cross-*session* and cross-*project* memory, or only per-procedure recall? The webinar's central promise rests on this. +- **Are skills necessary at all, or a 2025 scaffold?** [[thorsten-ball]] ships at the frontier without them. The corpus has never run the cheap test that would separate the readings — the same task, with and without a skill, in a non-engineer's hands. See [[skills-as-memory]]. +- Who pays for the **token budget** at *fleet* scale? *(Scoped 2026-07-28.)* For individuals the answer is settled and unremarkable — one subscription; the corpus's heaviest users ([[eugene]], 7 parallel agents on a $200 plan; [[allie-miller]], ~100 agents) report no ceiling, and the AI divide stays a **skill** gap. The open question is metered/fleet pricing and enterprise allocation — plus whether [[eugene]]'s price-*rise* prediction reopens it. See [[enterprise-ai-reality]]. +- If forms and admin panels die ([[build-for-the-agent-not-the-human]]), what does a non-technical person actually operate? "Prompt the agent" presumes exactly the competence the HR interviews identify as the bottleneck. diff --git a/wiki/queries/2026-07-22-webinar-theses.md b/wiki/queries/2026-07-22-webinar-theses.md index 3107aa6..37147cb 100644 --- a/wiki/queries/2026-07-22-webinar-theses.md +++ b/wiki/queries/2026-07-22-webinar-theses.md @@ -2,6 +2,14 @@ #query +> **⚠️ SUPERSEDED 2026-07-28 by [[2026-07-28-webinar-theses]].** Use that page. This one is preserved as the 2026-07-22 state of thinking (7 sources). +> +> What the refresh changed, in short: +> - **Thesis 3 ("Skills are the new memory") was reframed** to *context you author beats context that's inferred* — the v1 wording has a live counter-example ([[thorsten-ball]] ships 99%-AI-written code with no skills, no MCP, no slash commands) and the reframe is what all four practitioners actually agree on. See [[skills-as-memory]] for the three competing readings. +> - **Thesis 5 ("you build it, one small tool at a time") went from assertion to evidenced** via [[explosion-of-internal-software]]. +> - **Three theses added** that this set had no source for: verification / ask-for-checks, shedding weight, and ask-for-15-options. +> - Flagged by [[2026-07-28-lint]] as two ingests stale and effectively orphaned; the refresh is the fix. + ## Question "I need to make some theses for the webinar (theme: 'from chatbox to your own agentic operating system'). What theses can I suggest based on what you already have?" (2026-07-22) @@ -45,7 +53,7 @@ Grouped by the role they play in the talk. Each thesis is one sentence you could ## Evidence trail - [[overview]] — through-line and agree/diverge map -- Raw deliverables (not yet ingested, read directly): `raw/sources/Webinar Plan - From Chat Box to Your Own OS.md`, `raw/sources/Webinar script.md` (script ladder: Chat box → ReAct → Tools → Memory → Skills → Process → OS), `raw/sources/Ideas for webinar.md` +- Raw deliverables (authored, read directly; since moved to `raw/notes/`): `raw/notes/Webinar Plan - From Chat Box to Your Own OS.md`, `raw/notes/Webinar script.md` (script ladder: Chat box → ReAct → Tools → Memory → Skills → Process → OS), `raw/notes/Ideas for webinar.md` - Source summaries: [[2026-07-14-skills-based-on-git]], [[2026-07-14-gap-between-ai-users-irreversible]], [[2026-07-14-everything-we-knew-about-software-has-changed]], [[2026-07-14-sebastian-eugene-interview]], [[2026-07-14-nina-interview]], [[2026-07-14-yulia-interview]], [[2026-07-21-larysa-interview]] ## Follow-up questions diff --git a/wiki/queries/2026-07-24-non-engineer-throwaway-verification.md b/wiki/queries/2026-07-24-non-engineer-throwaway-verification.md index 820ded2..f3d0776 100644 --- a/wiki/queries/2026-07-24-non-engineer-throwaway-verification.md +++ b/wiki/queries/2026-07-24-non-engineer-throwaway-verification.md @@ -2,6 +2,8 @@ #query +> Carried forward as thesis **T6a** in [[2026-07-28-webinar-theses]] — "ask for checks, not just work." That refresh flags the absence of any verification beat in the current webinar script as the talk's biggest gap, since "can I trust it?" is the audience's first question. + **Question asked:** What is the non-engineer's analog of throwaway verification code ([[make-more-cheap-code]])? **Asked:** 2026-07-24 · **Status:** synthesis from existing pages (no new source) diff --git a/wiki/queries/2026-07-28-verification-beat-design.md b/wiki/queries/2026-07-28-verification-beat-design.md new file mode 100644 index 0000000..16a63e2 --- /dev/null +++ b/wiki/queries/2026-07-28-verification-beat-design.md @@ -0,0 +1,193 @@ +# Designing the Webinar's Verification Beat + +#query + +## Question + +"What are your suggestions for the verification beat? What can we add?" — following [[2026-07-28-webinar-theses]], which flagged the absence of any verification moment as the current script's biggest gap. (2026-07-28) + +## Answer — the headline + +**Make it one beat that closes both gaps, not two.** The refresh listed two holes in the script: no verification beat, and no "what stays yours" beat. They are the same beat. Verification is *precisely* where the human's remaining job lives ([[product-ownership]]), so a single moment can answer "can I trust it?" and "then what's left for me?" at once. In a 30-minute talk that matters — two gaps, one insertion. + +The frame that makes it land for non-engineers, from [[2026-07-24-non-engineer-throwaway-verification]]: + +> **You don't verify by reading everything. You verify by making a second, cheap, disposable agent try to break the first one.** +> Generated *checks*, not generated *content*. + +And the honest bottom line that becomes the "what stays yours" half: + +> A checker catches **drift**. It cannot catch a **wrong rule**. Mechanical correctness is delegable; judgment is not. + +## Placement — where the anxiety actually peaks + +The script's emotional arc is capability rising as the human recedes: *I do everything* → *I click a button*. The audience's unease peaks at one exact line in the **Process** station: + +> "I take my hands off the keyboard." … "Nobody is typing. It just... runs." + +That is the moment to answer it — not earlier (they don't feel it yet) and not in Q&A (too late). Recommended split: + +| Where | What | Cost | +|---|---|---| +| **Skills station** | Introduce the checker skill — the second species of skill | ~45–60 sec | +| **Process station** | Cash it in: the unattended process checks itself | ~20 sec (reuses an existing reveal) | +| **Closing arc** | The judgment half — what a checker *can't* do | ~30 sec, also fills gap #2 | + +--- + +## Option 1 — Minimal: one line in an artifact you already show *(≈15 seconds)* + +The Process station already reveals the prompt the shell wrote for its worker ("the AI wrote... a prompt. For another AI."). Add one visible line to that prompt: + +``` +After moving the cube, re-read the shelf state and confirm it matches the rule. +If it doesn't, report the mismatch instead of reporting success. +``` + +Then one spoken line: + +> Look at the last instruction it gave itself. +> +> "Check your own work — and if it's wrong, *say so* instead of saying done." +> +> It didn't just delegate the job. It delegated the *checking*. + +**Why this is the cheapest possible win:** the reveal already happens, the artifact is already on screen, and you add zero demo steps. If time is tight, do only this. + +--- + +## Option 2 — Recommended: the checker skill *(≈60–90 seconds)* + +Slots into the **Skills** station, immediately after `"save what we just did as a skill"`. The apparatus needed already exists: cube, shelves, weather widget, Override slider. + +**Draft copy, in the script's voice:** + +> So now I've got a skill that does the job. +> +> But here's the question you're all actually asking. +> +> If I'm not watching... how do I know it did it *right*? +> +> Let's give it a second skill. +> +> "save a skill that checks the first one — read the rule, read the weather, look at where the cube actually is, and tell me if they disagree" +> +> {AI writes `check-weather-based-movement/SKILL.md`} +> +> Two skills now. One does the job. +> +> One does nothing *but* look for the job being done wrong. +> +> Now watch — I'm going to break it on purpose. +> +> {drag the cube to the wrong shelf by hand} +> +> "run the check" +> +> {AI reports: rule says top shelf, cube is on bottom — mismatch} +> +> It caught it. +> +> And notice what that check cost me. +> +> One sentence. No code. It's a folder with a note in it — same as the first one. +> +> That's the trick nobody tells you about working with AI: +> +> you don't check the work by reading all of it. +> +> You check it by asking for something *cheap* whose only job is to find the mistake. + +**Why this specific demo works:** breaking it by hand is visible, instant, and unfakeable to a live audience — they see the cube in the wrong place *before* the agent says so. It also introduces the second species of skill (producers and checkers), which is a genuine corpus finding from [[2026-07-24-non-engineer-throwaway-verification]] and costs no new level. + +**Audience translation to say right after** — the demo is a cube, the takeaway must not be: + +> Same move, your work: +> "Read this job description as if you were a candidate who'd be put off by it — what did you see?" +> "Read this shortlist and argue *against* my top pick." +> +> That's not asking it to do the work. That's asking it to attack the work. + +--- + +## Option 3 — The ambiguity moment: you already wrote the perfect example *(≈30 seconds, standalone)* + +The Process station's goal line is: + +> "keep the cube on the **right shelf**: below 20 — top, above 20 — bottom. continuously." + +**"The right shelf"** is genuinely ambiguous — *correct* shelf, or the shelf on the *right*? The colon disambiguates it, so the script is safe as written. Which means you can deliberately show the unsafe version first: + +> Before I give it the real goal — watch this. +> +> {type only: "keep the cube on the right shelf"} +> +> {agent moves the cube to the right-hand shelf} +> +> That's not what I meant. +> +> I meant the *correct* shelf. It heard the shelf on the *right*. +> +> And here's the part that costs you: it didn't ask. It didn't hesitate. It just confidently did the wrong thing. +> +> {now type the full goal with the rule spelled out} +> +> Every gap you leave, it fills. And it fills it *silently*. + +This is the cheapest possible dramatization of [[leave-less-room-for-imagination]] — Eugene's sharpest claim, currently thesis T12 with no demo — and it doubles as verification motivation (*this* is what a checker catches). It costs one extra typed line and one cube movement. + +It also inverts cleanly, which is the durable insight from [[2026-07-24-non-engineer-throwaway-verification]]: **a fresh agent's misreading is a free ambiguity detector.** Before sending a brief to a human, hand it to a zero-context agent and ask what it thinks you meant. + +--- + +## The closing half — what a checker *can't* do *(fills gap #2)* + +The script's closing arc is currently all harness ("the model never changed… that's the harness"). Add the human half immediately before "You don't buy it. You build it": + +> One last thing — because I don't want to oversell this. +> +> That checker I wrote? It was written by the same AI it's checking. +> +> It'll catch the cube on the wrong shelf. Every time. +> +> What it will *never* catch... is Marcus's rule being wrong in the first place. +> +> If twenty degrees was the wrong number, both agents agree, confidently, forever. +> +> So here's the split, and it's the honest one: +> +> the machine checks whether the thing was done right. +> +> You check whether it was the right thing. +> +> That part doesn't get automated. That part is why you're still in the room. + +**Why this is worth the 30 seconds:** it is the strongest available answer to "will this replace me," it is honest rather than reassuring, and it converts the talk's ending from *capability* to *the audience's own value* — which is what an inspire talk should land on. + +## Recommended combination + +If you add **one** thing: Option 1 (15 sec, free). +If you add **one minute**: Option 2 + the closing half. +**Best value for ~2 minutes total:** Option 3 at Process → Option 2 at Skills → closing half. Option 3 creates the fear, Option 2 resolves it, the closing bounds the resolution honestly. + +Sequencing note: Option 3 sits *later* in the script than Option 2. If you use both, move the ambiguity moment earlier — into the Skills station just before the checker — so the problem precedes its solution. + +## Evidence trail + +- [[2026-07-24-non-engineer-throwaway-verification]] — the non-engineer analog: generated checks not content; checker skills as the second species; fresh-agent misread tests; the tier-D-stays-human caveat +- [[make-more-cheap-code]] — the engineer form (100:1 slop-to-ship), "have AI review before humans do," reading costs attention +- [[async-by-default]] — "you're async anyway, ask for proof," and the logged limit: *proof produced by the thing being checked is evidence, not verification* +- [[leave-less-room-for-imagination]] — drift's damage is what you don't notice; the source of Option 3 +- [[product-ownership]] — verification as the human's remaining craft; taste as the meta-skill +- [[2026-07-28-webinar-theses]] — T6a, and the two gaps this design closes +- Script state: `raw/notes/Webinar script.md` (Process and Skills stations, closing arc) + +## Open questions / honest caveats + +- **A checker written by the agent, checking the agent, is not independent.** It catches mechanical drift, not shared misunderstanding. The closing half says this out loud rather than hiding it — but if a technical audience member pushes, the real answer is that independence comes from *the human choosing the rule*, not from a second model. +- **Nina's finding is a standing counterweight:** transcript beat summary in her workflow ([[2026-07-14-nina-interview]]). A checker that reports "looks fine" is a summary. Don't let the beat imply reading is now optional — Theo's tier discipline is that some things still get read line by line. +- Untested: none of this has been run in front of a non-engineer audience. The cube demo may make verification feel mechanical in a way that doesn't transfer to judgment work — which is exactly why the audience-translation lines after Option 2 are load-bearing rather than optional. + +## Changed existing pages? + +No concept or entity pages changed — this is design synthesis on top of existing pages. `index.md` and `log.md` updated; [[2026-07-28-webinar-theses]] links here from T6a. diff --git a/wiki/queries/2026-07-28-webinar-theses.md b/wiki/queries/2026-07-28-webinar-theses.md new file mode 100644 index 0000000..7a1b6eb --- /dev/null +++ b/wiki/queries/2026-07-28-webinar-theses.md @@ -0,0 +1,156 @@ +# Webinar Theses v2 — From Chat Box to Your Own Agentic OS + +#query + +Supersedes [[2026-07-22-webinar-theses]] (7 sources). This set is synthesized from all **10** sources plus the current deliverable state in `raw/notes/` (`Webinar script.md`, `Webinar Plan - From Chat Box to Your Own OS.md`, `my theses.md`). + +## Question + +"Refresh the webinar theses" — restate the candidate theses for the talk *from chat box to your own agentic operating system*, now that [[2026-07-24-youre-reading-way-too-much-code]] and [[2026-07-28-agentic-engineering-10x-developer]] have been ingested. (2026-07-28) + +## What changed since v1 + +| | Change | +|---|---| +| **Strengthened** | Thesis 1 (harness not model) — Thorsten states the strongest form: *the dominant variable in output quality is the information you put in*, and he tells people to **stop tuning model choice**. This is now the best-evidenced claim in the vault and it is already the script's literal closing argument. | +| **Upgraded from assertion to evidence** | Thesis 5 (you build it, one small tool at a time) — [[explosion-of-internal-software]] supplies an outside, *non-engineer-shaped* case: a 20-person social club's ordering process encoded in **~2 hours of phone typing**, from a photo of a menu. The talk's least-provable claim is now its best-evidenced one. | +| **Weakened — needs reframing** | Thesis 3 (skills are the new memory) — a frontier practitioner ships 99%-AI-written code with **no skills, no MCP, no slash commands**. See T3 below for the reframe that survives him. | +| **New** | Three theses the earlier set had no source for: verification (T6a), shedding weight (T10), and variations-not-answers (T13a). | +| **New honest caveat** | **Token budget** — Thorsten names it as one of two winner/loser variables. The talk currently promises a skill gap can be closed by effort; this says part of it is closed by spending. | + +--- + +## The refreshed set + +Grouped by the job each does in the talk. Bold = recommended for the 30-min cut. + +> **Reading the numbers.** **T** = thesis; the numbers are stable handles so the theses can be referenced from other pages and in conversation ("T3 needs reframing") without re-quoting them. T1–T15 follow v1's order where the thesis survived, so a v1 number still points at roughly the same idea. A **letter suffix** (T6a, T13a) marks a thesis added in v2 next to its nearest relative rather than renumbering everything — T6a sits with T6 (both about where value goes when production is free), T13a with T13 (both method). Introduced in v2; v1 used plain 1–14. + +### Spine — what the talk claims + +**T1. The model isn't the product — the harness is.** *(strongest in the vault)* +Same model at every station; only the harness around it grows. Two independent frontier voices now say the same thing: the harness *is* the difference ([[harness]]), and "the dominant variable in output quality is the information you put in, not the model or the effort level" ([[thorsten-ball]]). Corollary you can say out loud: **stop shopping for models.** +— [[harness]], [[context-as-scarce-resource]] · script closing arc ("the model never changed") + +**T2. A chat box is an app you open; an OS is a system that runs around you.** +The rung ladder: stranger → doer → yours → teammate → knows you → always-on. Karpathy's framing (in the script's notes) is the same claim from outside: website → app you download → *self-contained, persistent, asynchronous entity working alongside teams*. +— [[levels-of-ai-usage]], [[personal-ai-operating-system]] · Plan through-line + +**T3. Context you *author* beats context that's *inferred*.** ⚠️ *reframed — see "The honest tension" below* +The v1 form was "skills are the new memory." That form now has a live counter-example. This reframe is what **all four** practitioners actually agree on: Konstantin's skills, Allie's foundation docs, Eugene's anti-memory position ("a skill is a file you can read, edit, version and delete"), *and* Thorsten's `AGENTS.md` are all the same move — context a human wrote on purpose, beating context a system guessed. It keeps the script's Memory→Skills stations intact while surviving the dissent. +— [[skills-as-memory]], [[personal-ai-operating-system]] · script Memory + Skills stations + +**T4. Context is the scarce resource — every rung is a technique for spending it wisely.** +Already dramatized in the script better than any slide could: *"the notebook is tiny. On purpose. Everything in it gets loaded into every single session — needed or not."* Then skills as two-stage loading: "the shelf can be huge — the desk stays clean." +— [[context-as-scarce-resource]], [[evolution-of-agent-tooling]] · script Memory→Skills transition + +**T5. You don't buy your OS — you build it, one small tool at a time.** *(now evidenced)* +Tools made for exactly one person, in an evening, asked-for rather than written. The new outside evidence matters because it defuses the obvious objection ("sure, *you* can do that — you're technical"): Thorsten's example is a social club, a phone, and a photo of a menu, and the software it replaced was a spreadsheet. +*(Scoped 2026-07-29: [[maintenance-is-the-real-cost]] adds the honest boundary — the cost of software is maintenance, not writing, and its pendulum case is a company abandoning its own Jira clone within four months. T5 survives because its examples pass that source's build-vs-buy checklist: tiny, personal, no users but you, no SLA. The claim is "little tools you make for yourself" — not "replace your vendors." Worth one sentence in the talk; it inoculates against the sharpest pushback a technical audience member could raise.)* +— [[explosion-of-internal-software]], [[emacsification-of-software]], [[personal-ai-operating-system]] · script OS section ("I didn't write it — I *asked* for it") + +### Stakes — why now + +**T6. The cost of producing work is going to zero; value migrates to directing and verifying it.** +Judgment, ownership, taste and relationships are what stay yours. Now has a hard datapoint: **99% of AMP's code is AI-written** — from inside a shipping company, not a demo. +— [[code-as-throwaway]], [[product-ownership]] + +**T6a. Verification is the new craft — and you get it by asking for checks, not just work.** *(new)* +The audience's real objection is "can I trust it?", and the current script has no answer. There is one: generate disposable work whose only job is to check the work you keep. Engineer form: 100 lines of slop verifying every shipped line ([[make-more-cheap-code]]). Non-engineer form: fresh-agent misread tests, parallel interpretations, checker skills, synthetic-candidate simulations ([[2026-07-24-non-engineer-throwaway-verification]]). Delegated form: *"you're async anyway — ask the agent for proof"* ([[async-by-default]]). +— **Designed in full at [[2026-07-28-verification-beat-design]]** (placement, drafted script copy, three options by cost). + +**T7. The gap between AI users and everyone else compounds — and is becoming irreversible.** +The person who builds their OS this week fears no release, because each capability slots into a system that already knows them. *(Thorsten names **token budget** as a second winner/loser variable, but that is a claim about metered agent-fleet work; for this audience the budget is one consumer subscription — keep the thesis on the skill gap. See [[enterprise-ai-reality]].)* +— [[2026-07-14-gap-between-ai-users-irreversible]] + +**T8. The more the world is mediated by AI proxies, the more valuable real human connection becomes.** +The "market of one" raises, not lowers, the price of being human. +— [[connections-as-moat]] + +### Obstacles — what the audience actually hits + +**T9. Adoption is blocked by friction, not resistance.** People aren't against AI — the setup is. Remove three clicks and they come. +— [[2026-07-14-nina-interview]], [[2026-07-14-yulia-interview]] + +**T10. Ask which of your processes only exist because *you* were the bottleneck.** *(new)* +Thorsten's knife, translated for a business audience: backlogs, status meetings, approval queues, the spreadsheet everyone re-keys. His test — *would this exist if agents had always been available?* His suggested exercise is a company-internal doc titled **"Software Is Dead — Now What?"**; the audience version is one honest list. Strong candidate for "Do this tonight." +— [[shedding-weight]] + +**T11. Even advanced users hit structural walls: no durable memory, integrations that dead-end, drift on loose specs.** +— [[2026-07-21-larysa-interview]], [[integration-dead-ends]] + +**T12. Leave less room for imagination.** Every gap in your instructions gets filled — invisibly. Now with a concrete, teachable structure instead of a principle: **set the standard → state intent → riff on the design → specify the process → set the constraints.** Thorsten's own gloss is the most quotable line for a non-technical crowd: *"This is how I would talk to a senior engineer. This is the Slack message I'd send."* +— [[leave-less-room-for-imagination]] + +### Method — what to do + +**T13. Solve first, then skillify.** Don't design up front — solve the task once in conversation, then freeze the working recipe. (~3 messages of correction, or >5 tool calls, = it's skill time.) The script already demonstrates this exactly: *"save what we just did as a skill."* +— [[solve-first-then-skillify]] + +**T13a. Ask for fifteen options, not one answer.** *(new)* +Generation is free, so the human's job moves from producing the artifact to **choosing among artifacts**. Thorsten's orb icon came from 15 AI-generated variants across 18 palettes; he picked one. This is the most immediately actionable thesis in the set for a non-technical audience — it needs no codebase, no skill, no setup — and it is the concrete form of "taste at AI speed." +— [[make-more-cheap-code]], [[product-ownership]] + +**T14. The assistant does the research; you do the judgment.** The Insights Collector meta-punchline: this talk was mined out of AI-processed interview notes. +— script §3 / Plan §3 + +**T15. Walk in a week what took the industry three years.** One hour of foundation docs · one skill from your #1 recurring annoyance · one real file tonight. +— [[levels-of-ai-usage]], [[personal-ai-operating-system]] + +--- + +## The honest tension (Q&A ammo — read this before you present) + +**Someone may ask whether skills are necessary at all.** They are right to. [[thorsten-ball]] runs a 99%-AI-written codebase with no skills, no MCP servers and no slash commands; his context lives in the codebase and a team-maintained `AGENTS.md`. The vault records three readings and settles none ([[skills-as-memory]]): + +1. **Situational** — he works daily in one codebase he controls, so his context can live in the code. Your audience has no codebase; skills are how a non-engineer gets the same effect. *(Strongest answer, and honest, but note he never scopes the claim himself.)* +2. **Premature abstraction** — skills are scaffolding for models that needed it, and a strong model plus a rich prompt may simply beat a skills library. +3. **Same thing under another name** — his `AGENTS.md` and AMP's curated sub-agents *are* two-stage authored context; the dispute is over who curates, not whether curation is needed. + +**The safe framing on stage is T3 as reframed above** — *authored beats inferred* — which is true under all three readings. Don't claim the skills mechanism is settled; it isn't, and the strongest counter-example is a frontier practitioner rather than a skeptic. + +Other live tensions, if the room is technical: +- Personal vs company-managed vs vendor-managed harness — [[enterprise-ai-reality]] +- Built-in agent memory as anti-feature (Eugene) vs persistent context docs used happily (Allie) — [[skills-as-memory]] +- Tight specs (T12) vs wide latitude — [[think-wider-not-bigger]] +- Model choice as a real lever (Eugene runs 4.7 over 4.8) vs a distraction (Thorsten) — [[leave-less-room-for-imagination]] + +--- + +## Recommended cut for the 30-min format + +Seven load-bearing theses, mapped to script beats. Changed from v1: **T3 reframed**, **T6a added**, T12 promoted (it now has a teachable structure), T9 demoted to Q&A. + +| # | Thesis | Script beat | +|---|---|---| +| T2 | App you open → system that runs around you | whole spine | +| T4 | Context is the scarce resource | Memory → Skills transition (already scripted) | +| T3 | Context you author beats context inferred | Memory + Skills stations | +| T6a | Ask for checks, not just work | **currently missing — see gaps** | +| T1 | The harness is the difference, not the model | closing arc (already scripted) | +| T5 | You don't buy it — you build it | OS section (already scripted) | +| T15 | Walk it in a week | "Do this tonight" | + +## Gaps in the current script this refresh exposes + +1. **No verification beat.** The script demonstrates capability at every station and never once shows the agent being *checked*. For an HR/BA audience whose first question is "can I trust it?", this is the biggest hole — and T6a fills it cheaply (one line in the Skills station: a second skill whose only job is to check the first). +2. **No "what stays yours" beat.** T6 and T8 are in the thesis set and in the Plan (§4), but the script's closing arc is entirely about the harness. The talk currently ends on capability, not on the human. + +*(A third gap — "the token-budget caveat is unsaid" — was proposed and **withdrawn 2026-07-28**. For this audience the budget is one consumer subscription, which is obvious and would land as a disclaimer. The corpus supports the withdrawal: [[eugene]] runs 7 parallel project-agents on a $200 plan and [[allie-miller]] runs ~100 agents, neither reporting a cost ceiling. Thorsten's token-budget variable describes metered agent-fleet work, not subscription use — see the scoping note on [[enterprise-ai-reality]].)* + +## Evidence trail + +- [[overview]] — through-line, agree/diverge map, vault-level open questions +- [[2026-07-28-lint]] — flagged v1 as two ingests stale and effectively orphaned; this page is the fix +- Sources: all 10, principally [[2026-07-28-agentic-engineering-10x-developer]], [[2026-07-24-youre-reading-way-too-much-code]], [[2026-07-14-skills-based-on-git]], [[2026-07-14-gap-between-ai-users-irreversible]], [[2026-07-21-larysa-interview]], [[2026-07-14-nina-interview]], [[2026-07-14-yulia-interview]], [[2026-07-14-sebastian-eugene-interview]] +- Deliverable state (raw, authored): `raw/notes/Webinar script.md` (ladder: Chat box → ReAct → Tools → Memory → Skills → Process → OS), `raw/notes/Webinar Plan - From Chat Box to Your Own OS.md`, `raw/notes/my theses.md` + +## Follow-up questions + +- Does T6a earn a station, or one line inside the Skills station? (Recommend the latter — a checker skill is one sentence of demo and costs no new level.) +- Should T10 ("which processes only exist because you were the bottleneck?") open the talk instead of closing it? It reframes the audience's own work before any capability is shown. +- The falsification test still unrun: same task, with and without a skill, in a non-engineer's hands. It would settle the T3 tension and would itself make a strong demo. + +## Changed existing pages? + +Yes — [[2026-07-22-webinar-theses]] marked superseded and pointed here. No concept or entity pages changed; this is synthesis. `index.md` and `log.md` updated. diff --git a/wiki/sources/2026-07-14-sebastian-eugene-interview.md b/wiki/sources/2026-07-14-sebastian-eugene-interview.md index 1b8c407..133bc7e 100644 --- a/wiki/sources/2026-07-14-sebastian-eugene-interview.md +++ b/wiki/sources/2026-07-14-sebastian-eugene-interview.md @@ -35,7 +35,7 @@ - **Entities:** [[sebastian]] · [[eugene]] · [[virtido]] · [[claude-code]] · [[hermes]] (Eugene references it among his tools) - **Concepts:** [[harness]] · [[enterprise-ai-reality]] · [[seniority-and-the-junior-squeeze]] · [[product-ownership]] · [[connections-as-moat]] · [[decoupling-identity-from-profession]] · [[code-as-throwaway]] - **Related sources:** [[2026-07-14-skills-based-on-git]] (harness definition + BYO-harness from the practitioner side) · [[2026-07-14-everything-we-knew-about-software-has-changed]] (identity baggage, code-as-throwaway) -- **Raw reference (not ingested):** `raw/sources/Ideas for webinar.md` echoes many of these (harness, connections, "describe problems not waterfalls," Daniel, HR search demo). +- **Raw reference (authored deliverable, not a source):** `raw/notes/Ideas for webinar.md` echoes many of these (harness, connections, "describe problems not waterfalls," Daniel, HR search demo). ## Open Questions diff --git a/wiki/sources/2026-07-28-agentic-engineering-10x-developer.md b/wiki/sources/2026-07-28-agentic-engineering-10x-developer.md new file mode 100644 index 0000000..66a7b65 --- /dev/null +++ b/wiki/sources/2026-07-28-agentic-engineering-10x-developer.md @@ -0,0 +1,62 @@ +# Agentic Engineering, explained by a 10x developer — Thorsten Ball + +#source + +## Source Metadata + +- **Date:** YouTube interview (publication date not stated in source), 42:33 — https://www.youtube.com/watch?v=FU5_kpTAVDo +- **Raw path:** `raw/sources/Agentic Engineering, explained by a 10x developer.md` +- **Source type:** podcast/video interview (conclusions doc) +- **Speaker:** [[thorsten-ball]] — founding engineer at [[amp]] (Sourcegraph); author of *Writing an Interpreter in Go* / *Writing a Compiler in Go* +- **Interviewer:** David Andre +- **Ingestion date:** 2026-07-28 + +## Core Claims + +- **The interesting variables moved.** Not "which model" or "how do I read every line," but: where does the information the agent needs live, how agent-friendly is your codebase/workflow, and what do you actually want to build. "The dominant variable in output quality is now the information you put in." +- **Shed weight.** Kill anything that only made sense before agents — backlogs, CI that re-runs the agent's own tests, IDE extensions, admin panels, local dev. AMP kills its own features publicly and calls itself "AMP Frontier Corporation." See [[shedding-weight]]. +- **99% of AMP is written by AI**, and that is compatible with high taste. "Most of slop comes from humans not having good product. With AI they can just build trash products faster." Slop = lack of ideas and playfulness, not an AI defect. +- **Build for the agent, not the human.** No human should fill out forms; anything a human can do on your site an agent should be able to do; ideally **bring your own agent**. See [[build-for-the-agent-not-the-human]]. +- **Software becomes bespoke** by two mechanisms: remixing existing software you don't upstream ([[emacsification-of-software]]) and building internal tools that used to be an Excel file ([[explosion-of-internal-software]]). +- **Async by default.** An *orb* — a remote sandbox tied to one conversation — packages thread + agent + computation + diff in one shareable URL. Delegate, do something else, and **ask for proof** because you're waiting anyway. See [[async-by-default]]. +- **He uses no skills, no MCP servers, no custom slash commands.** The only thing that matters to him is where the agent gets its information: training data (lossy, stale) plus the context window (your prompt, the codebase, `AGENTS.md`). *This directly contradicts the vault's skills spine — logged below and on [[skills-as-memory]].* +- **First-principles thinking is now the top skill.** "Everyone becomes an architect"; the value is seeing the workflow underneath the request and knowing solutions from other industries. + +## Key Evidence / Details + +- **Model choice:** once you have Fable 5 / GPT-5.6 Sol or equivalent, diminishing returns on which you pick, and on the effort level (medium vs high vs ultra). "If you're mad your model doesn't use camelCase, rethink your software engineering, not the model." AMP's default is medium. +- **AMP's setup:** installed as a PWA from `ampcode.com`; a low/medium/high/ultra dial mapping to model + sub-agent choice; sub-agents **Oracle** (reviewer/advisor) and **Painter** (images); meta-agent **Puck** that controls other agents, spawns orbs, messages them, runs flows. GPT, Anthropic and GLM models all supported. Agent-to-agent communication shipped "last Friday." +- **The hand-coding poll:** Thorsten polled the team with options 99%+, 90–99%, <90% — he didn't expect anyone under 80%. The one engineer who said "I still write a bunch by hand" landed at ~95% when pushed. +- **Taste at AI speed:** he had AI generate **15 versions** of the orb icon (Braille characters, different styles, 18 palettes) and picked one. AMP news imagery came from turn-by-turn Midjourney with two colleagues while reading Moby Dick. +- **The admin panel that dies:** he built a food-ordering app for a local club from a menu photo in 3 × 5-minute iterations; the agent also built an admin UI for prices and spelling. "I'm never going to open that. I'll just send another photo and say 'fix the pricing.'" A lot of admin UI existed only so that no code had to change. +- **The remix:** forked a diff viewer called **hunk**, told AMP "add Gruvbox dark hard theme, add file-checkoff in sidebar, compile, drop it in `~/bin`" — two minutes of agent time, no reason to upstream. +- **Internal software:** at his 20-person club he encoded the ordering process in ~2 hours of *phone typing*. Two variables separate winners from losers: knowing how to use agents, and **having the token budget** to do it. +- **The printer anti-example:** asked to build an app so a tablet prints a paper receipt for the kitchen, he pushed back — "Why do you need a printer? Why not a second tablet?" +- **Don't benchmark against the 1%:** Mitchell Hashimoto (Ghostty, GPU-accelerated terminal emulator) is cited in every online debate, but most software is CRUD, "MySQL and something-something," which agents handle fine. +- **His prompt structure** (porting Puck to the CLI): set the standard ("look at how it's implemented in web UI") → state intent → riff on the design → give explicit process ("research, document, sit down and think, compile what you learned, then come up with a good idea") → set sub-agent economics ("Fable is expensive, it scares me — use GPT models for the implementation"). "This is how I would talk to a senior engineer. This is the Slack message I'd send." +- **Velocity:** shipping velocity up in 4 weeks; designer Tim "never fixed so many paper cuts"; screenshot a bug → orb returns a fix → spot check → merge. A demo change was **shipped to production live during the podcast**. +- **Predictions:** local dev goes away; model distinctions matter less ("a button with which you can spawn a John Carmack"); unclear what software survives remixability; infra margins get eaten (15+ sandbox providers racing to zero); everyone moves up a level of abstraction. + +## Connections + +- **Entities:** [[thorsten-ball]] (new) · [[amp]] (new) · [[claude-code]] (peer harness) · [[theo-browne]] (closest ally in the corpus) +- **Concepts created:** [[shedding-weight]] · [[build-for-the-agent-not-the-human]] · [[emacsification-of-software]] · [[explosion-of-internal-software]] · [[async-by-default]] +- **Concepts reinforced:** [[context-as-scarce-resource]] (information > tuning; the token-budget variable) · [[code-as-throwaway]] (99% AI-written at a real company) · [[make-more-cheap-code]] (15 icon variations; ask for proof) · [[product-ownership]] (first-principles thinking, everyone an architect) · [[harness]] (AMP as a second reference harness; Oracle/Painter/Puck) · [[seniority-and-the-junior-squeeze]] (what took 2–3 years to teach is now a 30-second output) +- **Concepts contested:** [[skills-as-memory]] and [[evolution-of-agent-tooling]] (he skips the whole progression) · [[leave-less-room-for-imagination]] (model choice as a lever — he says stop tuning; Eugene selects 4.7 over 4.8) +- **Related sources:** [[2026-07-24-youre-reading-way-too-much-code]] and [[2026-07-14-everything-we-knew-about-software-has-changed]] (Theo — same strategic register, same slop-is-human stance); [[2026-07-14-sebastian-eugene-interview]] (the enterprise counterweight to "kill your process"); [[2026-07-21-larysa-interview]] (his `AGENTS.md`-only approach is exactly what leaves Larysa's memory gap unsolved) +- **Named but not in the vault:** Mitchell Hashimoto, Ghostty, hunk, Sourcegraph, Midjourney, Cloud9, Quinn (AMP CEO), Riverside + +## Open Questions + +- **Does "no skills, no MCP" generalize, or is it a property of his situation?** He works daily in one codebase he controls, with a company harness he helped build and a team-maintained `AGENTS.md`. The vault's skills case is strongest for people who move across many ad-hoc tasks and cannot encode context in a codebase (HR, BA). Neither side is tested against the other. Status: tentative. +- **What replaces the token budget as a constraint?** He names it as one of two winner/loser variables but says nothing about who pays for it — the missing economics of [[explosion-of-internal-software]]. +- All ratios (99% AI-written, the team poll) are self-reported from inside the company that sells the agent. Status: tentative. +- If admin panels and forms die, what does a non-technical person operate? Thorsten's answer is "prompt the agent," which assumes the prompting skill the corpus's HR interviews say is the actual bottleneck ([[levels-of-ai-usage]]). +- "Local dev is going away" is a prediction from a company selling remote sandboxes, and sits against [[eugene]]'s consolidated local workspace pitch ([[harness]]). Status: tentative. + +## Change Impact on Wiki + +- Created 5 concepts: [[shedding-weight]], [[build-for-the-agent-not-the-human]], [[emacsification-of-software]], [[explosion-of-internal-software]], [[async-by-default]]. +- Created 2 entities: [[thorsten-ball]], [[amp]]. +- Updated [[skills-as-memory]] and [[evolution-of-agent-tooling]] with the corpus's first credible *rejection* of the skills abstraction (recorded as a contradiction, not smoothed). +- Updated [[context-as-scarce-resource]] (information > tuning; token budget), [[code-as-throwaway]] (99%-AI-written datapoint; slop-is-human), [[make-more-cheap-code]] (variations-not-answers; ask-for-proof), [[product-ownership]] (first-principles as the top skill), [[harness]] (AMP; the minimal-harness position), [[seniority-and-the-junior-squeeze]] (the collapse of hand-taught senior knowledge), [[enterprise-ai-reality]] (token budget as a new access divide), [[leave-less-room-for-imagination]] (his prompt structure as a worked example; model-tuning tension), [[personal-ai-operating-system]] (the fourth layer — tools you build for yourself), [[claude-code]] (AMP as peer harness), [[theo-browne]] (ally cross-link), [[overview]] (9→10 sources; new "frontier side" of the through-line), `index.md`. diff --git a/wiki/sources/2026-07-29-what-if-we-vibe-code-it.md b/wiki/sources/2026-07-29-what-if-we-vibe-code-it.md new file mode 100644 index 0000000..6e1680a --- /dev/null +++ b/wiki/sources/2026-07-29-what-if-we-vibe-code-it.md @@ -0,0 +1,51 @@ +# А что если наВайб-Кодить? (What If We Vibe-Code It?) + +#source + +## Source Metadata + +- **Date of material:** 2026 (references tweets from March and July 2026) +- **Raw path:** `raw/sources/А что если наВайб-Кодить.md` +- **Source type:** viewer's conclusions from a YouTube video (5:31), Russian; https://www.youtube.com/watch?v=zBcWcignqng +- **Author:** unknown (a developer; his company uses Datadog and pays "literally millions a year" for it) +- **Ingestion date:** 2026-07-29 + +## Core Claims + +1. **Writing code was never the bottleneck — and never the cost.** Developers could always have written their own Jira or Datadog; they didn't because they didn't *want to run the result*. AI removes the writing cost, which was ~zero of the total, and leaves the real cost untouched. +2. **The cost of software is maintenance, not development.** The problem starts after the first user: bugs, regressions, feature requests, logs, monitoring, on-call, uptime responsibility. +3. **An internal service is an internal business.** A company that vibe-codes its own tracker/logger is either switching businesses or running two IT businesses at once — bad for the company (pays for one product, team builds another) and for the developer (two jobs, blamed for both). +4. **"I can write it in a week" ≠ "it's worth writing."** Between those two statements sit years of support. The vendor's price buys the removal of operational load, not the code. +5. **The pendulum case:** two tweets months apart — March 2026, a company builds its own Jira clone and migrates to it; July 2026, back to buying a tracker (Linear) because nobody wanted to carry the in-house product. "Assemble in two weeks — easy; carry it forward — impossible." +6. **A build-vs-buy checklist** for the AI era: size of the dependency (small, non-evolving libraries — fine to rewrite); does it need ongoing support (if yes, it's a separate project); what operational load does it add; is the business ready to open a second IT business inside itself. If the answers are bad, stay on the paid service — even with Claude / Antigravity / Codex at hand. +7. **Self-correction:** the author retracts his own earlier claim that "many services will die because of AI" — he now says he was wrong. + +## Key Evidence / Details + +- The Jira→Linear pendulum (claim 5) is the source's only external evidence; both tweets are second-hand and the company identification is fuzzy ("the same or a similar company"). Status: tentative. +- The author's own company is living the case study: pays millions/year for Datadog, is building an in-house replacement while also evaluating cheaper vendors — i.e. he criticizes the pattern from inside it, not from abstention. +- The checklist (claim 6) is prescriptive, not observed — the author's recommendation, not a documented practice. + +## Connections + +- Creates [[maintenance-is-the-real-cost]] — the source's central concept, new to the vault. +- Direct counterweight to [[explosion-of-internal-software]] and [[emacsification-of-software]]: both pages already flagged "maintenance is assumed away" as their weakest point; this source is the first to make that objection its whole thesis, with a named failure case. +- Sharpens the boundary of [[code-as-throwaway]] / [[make-more-cheap-code]]: throwaway code is safe *because it never has users*. The trap begins exactly where code stops being throwaway — the first user makes it a service. +- Agrees with the vault's spine from an unexpected angle: "writing code was never the bottleneck" is the same premise as [[harness]]-over-model and Theo's verification bottleneck — the sources disagree only about *which* non-writing cost dominates (verification vs. maintenance). +- Counters [[thorsten-ball]]'s prediction range: his club app passes the author's checklist (small, no SLA, no external users), but his "teams will remix Riverside" prediction is exactly what the pendulum case punishes. + +## Open Questions + +- Where is the threshold? The checklist says "small, non-evolving libraries — yes; services with users — no," but the interesting zone is between: a 20-person club app, an internal HR knowledge base, a personal fork in `~/bin`. +- Does the maintenance objection survive agents doing the maintenance? The author assumes ops load lands on humans; the corpus's outer-loop material ([[agentic-loops]], [[async-by-default]]) implies agents could carry some of it. Nobody in the corpus has evidence either way. +- Who is the author, and does his in-house Datadog replacement ship or die? The pendulum predicts die. + +## Change Impact on Wiki + +- Created [[maintenance-is-the-real-cost]] (new concept). +- [[explosion-of-internal-software]]: the "nobody owns the result" uncertainty upgraded from self-criticism to a sourced contradiction; scoping added (the club app passes the source's own checklist). +- [[emacsification-of-software]]: the "maintenance is assumed away" objection now has a source and a failure case. +- [[code-as-throwaway]]: boundary bullet added — throwaway is safe because unshipped; first user converts code into a service. +- [[thorsten-ball]]: contradiction added (pendulum case vs. his remix-and-build predictions). +- [[overview]]: divergence list and frontier-side bullet updated with the build-vs-buy counterweight. +- [[2026-07-28-webinar-theses]]: scoping note added to T5 (the thesis survives — its examples are checklist-safe — but gains an honest boundary).