From 25939d76c749f912392c8bfa2343c50f80960009 Mon Sep 17 00:00:00 2001 From: Booyaka101 Date: Fri, 7 Aug 2026 00:16:28 +0800 Subject: [PATCH 1/4] deps: move better-sqlite3 to 13.x, and drop the install script with it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Re-opens what #9 closed, with a corrected reading of why #9 failed. #9 was closed on the claim that 13.0.0-13.0.2 ship no prebuilt binaries. That was wrong: it was read off the GitHub *release* assets, which are empty for 13.x, not off the npm tarball, which is what npm actually installs. Registry metadata is unambiguous: 13.0.0 / 13.0.1 scripts.install = "node-gyp rebuild" gypfile = true 13.0.2 / 13.0.3 scripts.install = absent gypfile = false and the 13.0.2 and 13.0.3 tarballs both carry all 8 prebuilds, win32-x64 included. So 13.0.2 is already the fixed shape — the compile-from-source window was 13.0.0-13.0.1 only. What that does not yet explain is why #9's CI ran `node-gyp rebuild` for 13.0.2 and died on VS 2026 detection under node 22. I could not reproduce it: `npm ci` against #9's exact package.json + lockfile on Windows, node 22, under npm 10.9.3, 10.9.8 and 11 installs 44 packages in ~2s with no compile. Five clean attempts, no repro. This PR is the experiment that settles it — same bump, but with a lockfile regenerated by a real `npm install` rather than Dependabot's metadata-only rewrite, which is the one input I could not reproduce locally. If CI is green, that difference was the cause. Fallout of the move, all consistent with the install script being gone: allowScripts drops better-sqlite3 entirely, script-lens.json records zero packages with install-time behavior, and the lockfile loses 413 lines as the prebuild-install subtree goes with it. Offline tests pass. Cooldown: 13.0.3 clears the 72h window tomorrow. Co-Authored-By: Claude Opus 5 --- package-lock.json | 413 ++-------------------------------------------- package.json | 3 +- script-lens.json | 10 +- 3 files changed, 13 insertions(+), 413 deletions(-) diff --git a/package-lock.json b/package-lock.json index 0af7928..ecb52a9 100644 --- a/package-lock.json +++ b/package-lock.json @@ -10,7 +10,7 @@ "license": "MIT", "dependencies": { "@google/genai": "^2.12.0", - "better-sqlite3": "^12.11.1", + "better-sqlite3": "^13.0.3", "commander": "^15.0.0" }, "bin": { @@ -146,17 +146,15 @@ "license": "MIT" }, "node_modules/better-sqlite3": { - "version": "12.11.1", - "resolved": "https://registry.npmjs.org/better-sqlite3/-/better-sqlite3-12.11.1.tgz", - "integrity": "sha512-dq9AtApgg5PGFtBzPFSBl3HZQjHok5gaQCM6zh2Yk0aSmDCs1CbnVI8/HgASQkNKsWFpseIO9beg5xxpYhbIfA==", - "hasInstallScript": true, + "version": "13.0.3", + "resolved": "https://registry.npmjs.org/better-sqlite3/-/better-sqlite3-13.0.3.tgz", + "integrity": "sha512-RbOBxmLBG8uvFUc15X9+9SFemKcQ0WBuISBVkpuiaUB2qblC8UWlHEjdWVoZ8AdhSwmoEgsiXKfopX0CQxaACQ==", "license": "MIT", "dependencies": { - "bindings": "^1.5.0", - "prebuild-install": "^7.1.1" + "node-addon-api": "^8.0.0" }, "engines": { - "node": "20.x || 22.x || 23.x || 24.x || 25.x || 26.x" + "node": ">=22" } }, "node_modules/bignumber.js": { @@ -168,62 +166,12 @@ "node": "*" } }, - "node_modules/bindings": { - "version": "1.5.0", - "resolved": "https://registry.npmjs.org/bindings/-/bindings-1.5.0.tgz", - "integrity": "sha512-p2q/t/mhvuOj/UeLlV6566GD/guowlr0hHxClI0W9m7MWYkL1F0hLo+0Aexs9HSPCtR1SXQ0TD3MMKrXZajbiQ==", - "license": "MIT", - "dependencies": { - "file-uri-to-path": "1.0.0" - } - }, - "node_modules/bl": { - "version": "4.1.0", - "resolved": "https://registry.npmjs.org/bl/-/bl-4.1.0.tgz", - "integrity": "sha512-1W07cM9gS6DcLperZfFSj+bWLtaPGSOHWhPiGzXmvVJbRLdG82sH/Kn8EtW1VqWVA54AKf2h5k5BbnIbwF3h6w==", - "license": "MIT", - "dependencies": { - "buffer": "^5.5.0", - "inherits": "^2.0.4", - "readable-stream": "^3.4.0" - } - }, - "node_modules/buffer": { - "version": "5.7.1", - "resolved": "https://registry.npmjs.org/buffer/-/buffer-5.7.1.tgz", - "integrity": "sha512-EHcyIPBQ4BSGlvjB16k5KgAJ27CIsHY/2JBmCRReo48y9rQ3MaUzWX3KVlBa4U7MyX02HdVj0K7C3WaB3ju7FQ==", - "funding": [ - { - "type": "github", - "url": "https://github.com/sponsors/feross" - }, - { - "type": "patreon", - "url": "https://www.patreon.com/feross" - }, - { - "type": "consulting", - "url": "https://feross.org/support" - } - ], - "license": "MIT", - "dependencies": { - "base64-js": "^1.3.1", - "ieee754": "^1.1.13" - } - }, "node_modules/buffer-equal-constant-time": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/buffer-equal-constant-time/-/buffer-equal-constant-time-1.0.1.tgz", "integrity": "sha512-zRpUiDwd/xk6ADqPMATG8vc9VPrkck7T07OIx0gnjmJAnHnTVXNQG3vfvWNuiZIkwu9KrKdA1iJKfsfTVxE6NA==", "license": "BSD-3-Clause" }, - "node_modules/chownr": { - "version": "1.1.4", - "resolved": "https://registry.npmjs.org/chownr/-/chownr-1.1.4.tgz", - "integrity": "sha512-jJ0bqzaylmJtVnNgzTeSOs8DPavpbYgEr/b0YL8/2GO3xJEhInFmhKMUnEJQjZumK7KXGFhUy89PrsJWlakBVg==", - "license": "ISC" - }, "node_modules/commander": { "version": "15.0.0", "resolved": "https://registry.npmjs.org/commander/-/commander-15.0.0.tgz", @@ -259,39 +207,6 @@ } } }, - "node_modules/decompress-response": { - "version": "6.0.0", - "resolved": "https://registry.npmjs.org/decompress-response/-/decompress-response-6.0.0.tgz", - "integrity": "sha512-aW35yZM6Bb/4oJlZncMH2LCoZtJXTRxES17vE3hoRiowU2kWHaJKFkSBDnDR+cm9J+9QhXmREyIfv0pji9ejCQ==", - "license": "MIT", - "dependencies": { - "mimic-response": "^3.1.0" - }, - "engines": { - "node": ">=10" - }, - "funding": { - "url": "https://github.com/sponsors/sindresorhus" - } - }, - "node_modules/deep-extend": { - "version": "0.6.0", - "resolved": "https://registry.npmjs.org/deep-extend/-/deep-extend-0.6.0.tgz", - "integrity": "sha512-LOHxIOaPYdHlJRtCQfDIVZtfw/ufM8+rVj649RIHzcm/vGwQRXFt6OPqIFWsm2XEMrNIEtWR64sY1LEKD2vAOA==", - "license": "MIT", - "engines": { - "node": ">=4.0.0" - } - }, - "node_modules/detect-libc": { - "version": "2.1.2", - "resolved": "https://registry.npmjs.org/detect-libc/-/detect-libc-2.1.2.tgz", - "integrity": "sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==", - "license": "Apache-2.0", - "engines": { - "node": ">=8" - } - }, "node_modules/ecdsa-sig-formatter": { "version": "1.0.11", "resolved": "https://registry.npmjs.org/ecdsa-sig-formatter/-/ecdsa-sig-formatter-1.0.11.tgz", @@ -301,24 +216,6 @@ "safe-buffer": "^5.0.1" } }, - "node_modules/end-of-stream": { - "version": "1.4.5", - "resolved": "https://registry.npmjs.org/end-of-stream/-/end-of-stream-1.4.5.tgz", - "integrity": "sha512-ooEGc6HP26xXq/N+GCGOT0JKCLDGrq2bQUZrQ7gyrJiZANJ/8YDTxTpQBXGMn+WbIQXNVpyWymm7KYVICQnyOg==", - "license": "MIT", - "dependencies": { - "once": "^1.4.0" - } - }, - "node_modules/expand-template": { - "version": "2.0.3", - "resolved": "https://registry.npmjs.org/expand-template/-/expand-template-2.0.3.tgz", - "integrity": "sha512-XYfuKMvj4O35f/pOXLObndIRvyQ+/+6AhODh+OKWj9S9498pHHn/IMszH+gt0fBCRWMNfk1ZSp5x3AifmnI2vg==", - "license": "(MIT OR WTFPL)", - "engines": { - "node": ">=6" - } - }, "node_modules/extend": { "version": "3.0.2", "resolved": "https://registry.npmjs.org/extend/-/extend-3.0.2.tgz", @@ -348,12 +245,6 @@ "node": "^12.20 || >= 14.13" } }, - "node_modules/file-uri-to-path": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/file-uri-to-path/-/file-uri-to-path-1.0.0.tgz", - "integrity": "sha512-0Zt+s3L7Vf1biwWZ29aARiVYLx7iMGnEUl9x33fbB/j3jR81u/O2LbqK+Bm1CDSNDKVtJ/YjwY7TUd5SkeLQLw==", - "license": "MIT" - }, "node_modules/formdata-polyfill": { "version": "4.0.10", "resolved": "https://registry.npmjs.org/formdata-polyfill/-/formdata-polyfill-4.0.10.tgz", @@ -366,12 +257,6 @@ "node": ">=12.20.0" } }, - "node_modules/fs-constants": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/fs-constants/-/fs-constants-1.0.0.tgz", - "integrity": "sha512-y6OAwoSIf7FyjMIv94u+b5rdheZEjzR63GTyZJm5qh4Bi+2YgwLCcI/fPFZkL5PSixOt6ZNKm+w+Hfp/Bciwow==", - "license": "MIT" - }, "node_modules/gaxios": { "version": "7.2.0", "resolved": "https://registry.npmjs.org/gaxios/-/gaxios-7.2.0.tgz", @@ -400,12 +285,6 @@ "node": ">=18" } }, - "node_modules/github-from-package": { - "version": "0.0.0", - "resolved": "https://registry.npmjs.org/github-from-package/-/github-from-package-0.0.0.tgz", - "integrity": "sha512-SyHy3T1v2NUXn29OsWdxmK6RwHD+vkj3v8en8AOBZ1wBQ/hCAQ5bAQTD02kW4W9tUp/3Qh6J8r9EvntiyCmOOw==", - "license": "MIT" - }, "node_modules/google-auth-library": { "version": "10.9.0", "resolved": "https://registry.npmjs.org/google-auth-library/-/google-auth-library-10.9.0.tgz", @@ -445,38 +324,6 @@ "node": ">= 14" } }, - "node_modules/ieee754": { - "version": "1.2.1", - "resolved": "https://registry.npmjs.org/ieee754/-/ieee754-1.2.1.tgz", - "integrity": "sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA==", - "funding": [ - { - "type": "github", - "url": "https://github.com/sponsors/feross" - }, - { - "type": "patreon", - "url": "https://www.patreon.com/feross" - }, - { - "type": "consulting", - "url": "https://feross.org/support" - } - ], - "license": "BSD-3-Clause" - }, - "node_modules/inherits": { - "version": "2.0.4", - "resolved": "https://registry.npmjs.org/inherits/-/inherits-2.0.4.tgz", - "integrity": "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ==", - "license": "ISC" - }, - "node_modules/ini": { - "version": "1.3.8", - "resolved": "https://registry.npmjs.org/ini/-/ini-1.3.8.tgz", - "integrity": "sha512-JV/yugV2uzW5iMRSiZAyDtQd+nxtUnjeLt0acNdw98kKLrvuRVyB80tsREOE7yvGVgalhZ6RNXCmEHkUKBKxew==", - "license": "ISC" - }, "node_modules/json-bigint": { "version": "1.0.0", "resolved": "https://registry.npmjs.org/json-bigint/-/json-bigint-1.0.0.tgz", @@ -513,55 +360,19 @@ "integrity": "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==", "license": "Apache-2.0" }, - "node_modules/mimic-response": { - "version": "3.1.0", - "resolved": "https://registry.npmjs.org/mimic-response/-/mimic-response-3.1.0.tgz", - "integrity": "sha512-z0yWI+4FDrrweS8Zmt4Ej5HdJmky15+L2e6Wgn3+iK5fWzb6T3fhNFq2+MeTRb064c6Wr4N/wv0DzQTjNzHNGQ==", - "license": "MIT", - "engines": { - "node": ">=10" - }, - "funding": { - "url": "https://github.com/sponsors/sindresorhus" - } - }, - "node_modules/minimist": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/minimist/-/minimist-1.2.8.tgz", - "integrity": "sha512-2yyAR8qBkN3YuheJanUpWC5U3bb5osDywNB8RzDVlDwDHbocAJveqqj1u8+SVD7jkWT4yvsHCpWqqWqAxb0zCA==", - "license": "MIT", - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/mkdirp-classic": { - "version": "0.5.3", - "resolved": "https://registry.npmjs.org/mkdirp-classic/-/mkdirp-classic-0.5.3.tgz", - "integrity": "sha512-gKLcREMhtuZRwRAfqP3RFW+TK4JqApVBtOIftVgjuABpAtpxhPGaDcfvbhNvD0B8iD1oUr/txX35NjcaY6Ns/A==", - "license": "MIT" - }, "node_modules/ms": { "version": "2.1.3", "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", "integrity": "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==", "license": "MIT" }, - "node_modules/napi-build-utils": { - "version": "2.0.0", - "resolved": "https://registry.npmjs.org/napi-build-utils/-/napi-build-utils-2.0.0.tgz", - "integrity": "sha512-GEbrYkbfF7MoNaoh2iGG84Mnf/WZfB0GdGEsM8wz7Expx/LlWf5U8t9nvJKXSp3qr5IsEbK04cBGhol/KwOsWA==", - "license": "MIT" - }, - "node_modules/node-abi": { - "version": "3.94.0", - "resolved": "https://registry.npmjs.org/node-abi/-/node-abi-3.94.0.tgz", - "integrity": "sha512-W5ZNO5KRPB5TkYmGVD9F6YqhsglXJzE6etpbmT+f6EQElhiX/UTG551cnsRGvLG3fyZEg9HwaDmNmj5nwJ4z9g==", + "node_modules/node-addon-api": { + "version": "8.9.1", + "resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.9.1.tgz", + "integrity": "sha512-4eUQWVPCUUUiBjLnHS3cXWeC6ryoPUc0U3rP7IuzapoGbzMqd/r6KKO0clr0b+snQhsrueFEhCZDdK+LK7hxKg==", "license": "MIT", - "dependencies": { - "semver": "^7.3.5" - }, "engines": { - "node": ">=10" + "node": "^18 || ^20 || >= 21" } }, "node_modules/node-domexception": { @@ -602,15 +413,6 @@ "url": "https://opencollective.com/node-fetch" } }, - "node_modules/once": { - "version": "1.4.0", - "resolved": "https://registry.npmjs.org/once/-/once-1.4.0.tgz", - "integrity": "sha512-lNaJgI+2Q5URQBkccEKHTQOPaXdUxnZZElQTZY0MFUAuaEqe1E+Nyvgdz/aIyNi6Z9MzO5dv1H8n58/GELp3+w==", - "license": "ISC", - "dependencies": { - "wrappy": "1" - } - }, "node_modules/p-retry": { "version": "4.6.2", "resolved": "https://registry.npmjs.org/p-retry/-/p-retry-4.6.2.tgz", @@ -624,33 +426,6 @@ "node": ">=8" } }, - "node_modules/prebuild-install": { - "version": "7.1.3", - "resolved": "https://registry.npmjs.org/prebuild-install/-/prebuild-install-7.1.3.tgz", - "integrity": "sha512-8Mf2cbV7x1cXPUILADGI3wuhfqWvtiLA1iclTDbFRZkgRQS0NqsPZphna9V+HyTEadheuPmjaJMsbzKQFOzLug==", - "deprecated": "No longer maintained. Please contact the author of the relevant native addon; alternatives are available.", - "license": "MIT", - "dependencies": { - "detect-libc": "^2.0.0", - "expand-template": "^2.0.3", - "github-from-package": "0.0.0", - "minimist": "^1.2.3", - "mkdirp-classic": "^0.5.3", - "napi-build-utils": "^2.0.0", - "node-abi": "^3.3.0", - "pump": "^3.0.0", - "rc": "^1.2.7", - "simple-get": "^4.0.0", - "tar-fs": "^2.0.0", - "tunnel-agent": "^0.6.0" - }, - "bin": { - "prebuild-install": "bin.js" - }, - "engines": { - "node": ">=10" - } - }, "node_modules/protobufjs": { "version": "7.6.5", "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.5.tgz", @@ -674,45 +449,6 @@ "node": ">=12.0.0" } }, - "node_modules/pump": { - "version": "3.0.4", - "resolved": "https://registry.npmjs.org/pump/-/pump-3.0.4.tgz", - "integrity": "sha512-VS7sjc6KR7e1ukRFhQSY5LM2uBWAUPiOPa/A3mkKmiMwSmRFUITt0xuj+/lesgnCv+dPIEYlkzrcyXgquIHMcA==", - "license": "MIT", - "dependencies": { - "end-of-stream": "^1.1.0", - "once": "^1.3.1" - } - }, - "node_modules/rc": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/rc/-/rc-1.2.8.tgz", - "integrity": "sha512-y3bGgqKj3QBdxLbLkomlohkvsA8gdAiUQlSBJnBhfn+BPxg4bc62d8TcBW15wavDfgexCgccckhcZvywyQYPOw==", - "license": "(BSD-2-Clause OR MIT OR Apache-2.0)", - "dependencies": { - "deep-extend": "^0.6.0", - "ini": "~1.3.0", - "minimist": "^1.2.0", - "strip-json-comments": "~2.0.1" - }, - "bin": { - "rc": "cli.js" - } - }, - "node_modules/readable-stream": { - "version": "3.6.2", - "resolved": "https://registry.npmjs.org/readable-stream/-/readable-stream-3.6.2.tgz", - "integrity": "sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA==", - "license": "MIT", - "dependencies": { - "inherits": "^2.0.3", - "string_decoder": "^1.1.1", - "util-deprecate": "^1.0.1" - }, - "engines": { - "node": ">= 6" - } - }, "node_modules/retry": { "version": "0.13.1", "resolved": "https://registry.npmjs.org/retry/-/retry-0.13.1.tgz", @@ -742,133 +478,12 @@ ], "license": "MIT" }, - "node_modules/semver": { - "version": "7.8.5", - "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.5.tgz", - "integrity": "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA==", - "license": "ISC", - "bin": { - "semver": "bin/semver.js" - }, - "engines": { - "node": ">=10" - } - }, - "node_modules/simple-concat": { - "version": "1.0.1", - "resolved": "https://registry.npmjs.org/simple-concat/-/simple-concat-1.0.1.tgz", - "integrity": "sha512-cSFtAPtRhljv69IK0hTVZQ+OfE9nePi/rtJmw5UjHeVyVroEqJXP1sFztKUy1qU+xvz3u/sfYJLa947b7nAN2Q==", - "funding": [ - { - "type": "github", - "url": "https://github.com/sponsors/feross" - }, - { - "type": "patreon", - "url": "https://www.patreon.com/feross" - }, - { - "type": "consulting", - "url": "https://feross.org/support" - } - ], - "license": "MIT" - }, - "node_modules/simple-get": { - "version": "4.0.1", - "resolved": "https://registry.npmjs.org/simple-get/-/simple-get-4.0.1.tgz", - "integrity": "sha512-brv7p5WgH0jmQJr1ZDDfKDOSeWWg+OVypG99A/5vYGPqJ6pxiaHLy8nxtFjBA7oMa01ebA9gfh1uMCFqOuXxvA==", - "funding": [ - { - "type": "github", - "url": "https://github.com/sponsors/feross" - }, - { - "type": "patreon", - "url": "https://www.patreon.com/feross" - }, - { - "type": "consulting", - "url": "https://feross.org/support" - } - ], - "license": "MIT", - "dependencies": { - "decompress-response": "^6.0.0", - "once": "^1.3.1", - "simple-concat": "^1.0.0" - } - }, - "node_modules/string_decoder": { - "version": "1.3.0", - "resolved": "https://registry.npmjs.org/string_decoder/-/string_decoder-1.3.0.tgz", - "integrity": "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA==", - "license": "MIT", - "dependencies": { - "safe-buffer": "~5.2.0" - } - }, - "node_modules/strip-json-comments": { - "version": "2.0.1", - "resolved": "https://registry.npmjs.org/strip-json-comments/-/strip-json-comments-2.0.1.tgz", - "integrity": "sha512-4gB8na07fecVVkOI6Rs4e7T6NOTki5EmL7TUduTs6bu3EdnSycntVJ4re8kgZA+wx9IueI2Y11bfbgwtzuE0KQ==", - "license": "MIT", - "engines": { - "node": ">=0.10.0" - } - }, - "node_modules/tar-fs": { - "version": "2.1.5", - "resolved": "https://registry.npmjs.org/tar-fs/-/tar-fs-2.1.5.tgz", - "integrity": "sha512-OboTd8mmMhZDNPV+UjQcK9yKAatXu2aJ+r1w4im1Otd4M4fl2hwvdoXUxIYHFTHWK/3y3FarBP70v3vwmGlOxw==", - "license": "MIT", - "dependencies": { - "chownr": "^1.1.1", - "mkdirp-classic": "^0.5.2", - "pump": "^3.0.0", - "tar-stream": "^2.1.4" - } - }, - "node_modules/tar-stream": { - "version": "2.2.0", - "resolved": "https://registry.npmjs.org/tar-stream/-/tar-stream-2.2.0.tgz", - "integrity": "sha512-ujeqbceABgwMZxEJnk2HDY2DlnUZ+9oEcb1KzTVfYHio0UE6dG71n60d8D2I4qNvleWrrXpmjpt7vZeF1LnMZQ==", - "license": "MIT", - "dependencies": { - "bl": "^4.0.3", - "end-of-stream": "^1.4.1", - "fs-constants": "^1.0.0", - "inherits": "^2.0.3", - "readable-stream": "^3.1.1" - }, - "engines": { - "node": ">=6" - } - }, - "node_modules/tunnel-agent": { - "version": "0.6.0", - "resolved": "https://registry.npmjs.org/tunnel-agent/-/tunnel-agent-0.6.0.tgz", - "integrity": "sha512-McnNiV1l8RYeY8tBgEpuodCC1mLUdbSN+CYBL7kJsJNInOP8UjDDEwdk6Mw60vdLLrr5NHKZhMAOSrR2NZuQ+w==", - "license": "Apache-2.0", - "dependencies": { - "safe-buffer": "^5.0.1" - }, - "engines": { - "node": "*" - } - }, "node_modules/undici-types": { "version": "8.3.0", "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz", "integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==", "license": "MIT" }, - "node_modules/util-deprecate": { - "version": "1.0.2", - "resolved": "https://registry.npmjs.org/util-deprecate/-/util-deprecate-1.0.2.tgz", - "integrity": "sha512-EPD5q1uXyFxJpCrLnCc1nHnq3gOa6DZBocAIiI2TaSCA7VCJ1UJDMagCzIkXNsUYfD1daK//LTEQ8xiIbrHtcw==", - "license": "MIT" - }, "node_modules/web-streams-polyfill": { "version": "3.3.3", "resolved": "https://registry.npmjs.org/web-streams-polyfill/-/web-streams-polyfill-3.3.3.tgz", @@ -878,12 +493,6 @@ "node": ">= 8" } }, - "node_modules/wrappy": { - "version": "1.0.2", - "resolved": "https://registry.npmjs.org/wrappy/-/wrappy-1.0.2.tgz", - "integrity": "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==", - "license": "ISC" - }, "node_modules/ws": { "version": "8.21.1", "resolved": "https://registry.npmjs.org/ws/-/ws-8.21.1.tgz", diff --git a/package.json b/package.json index d7dc24f..eebef17 100644 --- a/package.json +++ b/package.json @@ -46,12 +46,11 @@ "homepage": "https://github.com/Booyaka101/gemcatch#readme", "dependencies": { "@google/genai": "^2.12.0", - "better-sqlite3": "^12.11.1", + "better-sqlite3": "^13.0.3", "commander": "^15.0.0" }, "allowScripts": { "@google/genai@2.15.0": true, - "better-sqlite3@12.11.1": true, "protobufjs@7.6.5": true } } diff --git a/script-lens.json b/script-lens.json index efb66de..75e522d 100644 --- a/script-lens.json +++ b/script-lens.json @@ -1,13 +1,5 @@ { "tool": "npm-script-lens", "version": "1.4.0", - "packages": { - "better-sqlite3@12.11.1": { - "risk": "HIGH", - "capabilities": [ - "exec", - "gyp" - ] - } - } + "packages": {} } From c5869cc852c0493608a606da1ed5b3f1be3e2239 Mon Sep 17 00:00:00 2001 From: Booyaka101 Date: Sat, 8 Aug 2026 09:02:18 +0800 Subject: [PATCH 2/4] feat: Deep Research agents, spend guard, citations; default model to 3.5-flash-lite (0.4.0) - research/batch grow -a/--agent: interactions.create is sent `agent` INSTEAD of `model` (mutually exclusive; passing both is a clean error). Aliases resolve through one table (deep-research, deep-research-max -> the full preview ids); unknown ids pass through so future agents need no release. - Spend guard: agent submissions print the documented per-task band ($1-$3 / $3-$7, quoted with the docs own hedge), batch prints N x band, and require y/N confirmation -- --yes when stdin is not a TTY, --dry-run to preview. Declining or refusing writes zero rows. - Result extraction takes the final answer-bearing step (docs: steps[-1].content[0].text) with a collect-everything fallback; citations are persisted, printed as Sources:, and carried in --json. - Additive migration: agent + citations columns; pre-0.4.0 stores upgrade in place, agent NULL. list grows an AGENT column when relevant; stats tallies agent runs. - Default model: gemini-3.1-flash-lite -> gemini-3.5-flash-lite (GA 2026-07-21). - 15 new offline tests (agent shapes, guard paths, budget-pause incomplete, agent 404 retirement, v0.3.0 migration); all 49 pre-existing tests intact. Co-Authored-By: Claude Fable 5 --- CHANGELOG.md | 45 ++++++- CONTRIBUTING.md | 3 + README.md | 50 +++++++- db.js | 21 +++- gemini.js | 113 +++++++++++++++-- index.js | 185 ++++++++++++++++++++++++---- package.json | 3 +- test-offline.js | 319 +++++++++++++++++++++++++++++++++++++++++++++++- 8 files changed, 693 insertions(+), 46 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3e1e220..db1c465 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,48 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +## [0.4.0] - 2026-08-08 + +### Added + +- **Research agents.** `-a, --agent ` on `research` and `batch` submits to a + Gemini Deep Research agent instead of a model — `interactions.create` is sent + `agent` *instead of* `model` (they are mutually exclusive, and passing both is + a clean error). Aliases resolve through one table: `deep-research` → + `deep-research-preview-04-2026`, `deep-research-max` → + `deep-research-max-preview-04-2026`; any other value passes through unchanged, + so a future agent id works without a gemcatch release. Agents *require* + background execution, which gemcatch has always set — and on the free tier the + finished report is dropped after 1 day, which is exactly the race the daemon + exists to win. The agent is recorded per task, shown in `list` (the AGENT + column appears when a listing contains agent runs) and tallied in `stats`. +- **Spend guard.** Deep Research is documented at $1.00–$3.00 per task and Deep + Research Max at $3.00–$7.00 (estimates based on preview rates, per the docs, + and subject to change). Every agent submission prints its band first — + `batch` prints N × the band as a total — and asks for an interactive `y/N` + confirmation. When stdin is not a TTY, `--yes` is required and anything else + is refused before a row is written; declining writes nothing and exits + non-zero. `--dry-run` (now on `research` too) prints the full projected spend + and submits nothing. +- **Citations.** Agent runs return citations alongside the report; the docs say + to review them to verify the sources, so they are persisted (new `citations` + column, JSON) rather than discarded, printed under the result as a `Sources:` + list, and carried in `--json` output. +- Result extraction now takes the **final answer-bearing step** — where the + docs place an agent's completed report (`steps[-1].content[0].text`) and + where a model run's `model_output` already sits — with a fall-back to the old + collect-everything behaviour if that step carries no text, so an unexpected + shape can never silently blank a result. No special-casing on the agent id. +- Additive schema migration: `agent` and `citations` columns. A pre-0.4.0 + `tasks.db` upgrades in place, keeps every row, and reports `agent` as NULL + for them. + +### Changed + +- The default model is now **`gemini-3.5-flash-lite`** (GA on 2026-07-21), + replacing the older `gemini-3.1-flash-lite`. Override with `GEMCATCH_MODEL` + or `--model` as before. + ## [0.3.0] - 2026-07-19 ### Added @@ -162,7 +204,8 @@ seen a task complete, the text is cached locally and survives that expiry — bu something has to poll inside that window for it to be seen at all, which is what `gemcatch daemon` exists to do. -[Unreleased]: https://github.com/Booyaka101/gemcatch/compare/v0.3.0...HEAD +[Unreleased]: https://github.com/Booyaka101/gemcatch/compare/v0.4.0...HEAD +[0.4.0]: https://github.com/Booyaka101/gemcatch/compare/v0.3.0...v0.4.0 [0.3.0]: https://github.com/Booyaka101/gemcatch/compare/v0.2.0...v0.3.0 [0.2.0]: https://github.com/Booyaka101/gemcatch/compare/v0.1.1...v0.2.0 [0.1.1]: https://github.com/Booyaka101/gemcatch/compare/v0.1.0...v0.1.1 diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f3716e8..58b84da 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -48,6 +48,9 @@ These are all load-bearing and were each learned the hard way: - **Every call goes through `call()`**, which paces it (`gate()`) and retries it. Add a new API operation and it must too, or it silently escapes both. - **Concurrency is not a rate.** `mapLimit` in `index.js` bounds how many polls are open at once; `GEMCATCH_RPM` in `gemini.js` is what actually keeps a wide fan-out inside the free tier's requests-per-minute allowance. They are different limits and both matter. - **Only transient failures retry.** 408/429/5xx and network errors, never 4xx: a bad key or bad model id fails the same way forever, so retrying it just spends the user's quota to reach the identical error. `shouldRetry()` is the one place that decides, and it's pinned by a test. +- **`agent` replaces `model` on create — they are mutually exclusive.** An agent run is submitted with `agent` and no `model`; the CLI rejects the combination before anything is written. The full preview agent ids live in ONE table (`AGENT_ALIASES` in `gemini.js`) — never hardcode them at a call site, they will be superseded. +- **An agent's report is in the FINAL step** (`steps[-1].content[0].text` per the docs); the earlier steps are its plan and interim drafts. `textFromSteps` takes the last answer-bearing step for every run — no special-casing on the agent id — and falls back to collecting everything if that step has no text. +- **Agent submissions cost dollars per task, so the spend guard is load-bearing.** Any new path that reaches `gemini.submit` with an agent must go through `confirmSpend()` first, *before* any row is written — a declined confirmation must leave the store untouched. `GEMCATCH_ASSUME_TTY=1` is the test hook that lets the suite drive the interactive y/N branch through a pipe. ## Tests diff --git a/README.md b/README.md index 7124c16..71a064a 100644 --- a/README.md +++ b/README.md @@ -24,7 +24,7 @@ The EU AI Act's high-risk obligations phase in from August 2026, whereas... ## Setup -Needs Node.js 22+ and a Gemini API key. **Getting a key needs no billing account and no card.** `gemini-3.1-flash-lite` runs free within the [free tier's](https://ai.google.dev/gemini-api/docs/pricing) daily quota; past that, paid rates apply. +Needs Node.js 22+ and a Gemini API key. **Getting a key needs no billing account and no card.** `gemini-3.5-flash-lite` (the default model, GA since July 2026) runs free within the [free tier's](https://ai.google.dev/gemini-api/docs/pricing) daily quota; past that, paid rates apply. 1. Get a key at **** 2. Put it in your environment: @@ -99,6 +99,8 @@ Useful flags: | --- | --- | --- | | `--json` | most commands | Machine-readable output. | | `-m, --model ` | `research`, `batch` | Override the model. | +| `-a, --agent ` | `research`, `batch` | Submit to a [research agent](#research-agents) instead of a model. Mutually exclusive with `--model`. | +| `--yes` | `research`, `batch` | Confirm the agent cost without asking. Required for `--agent` when stdin is not a TTY. | | `-s, --system ` | `research`, `batch` | Set a system instruction. | | `-f, --file ` | `research` | Read the prompt from a file. | | `-t, --tag ` | `research`, `batch`, `list` | Label tasks and filter them. | @@ -110,7 +112,7 @@ Useful flags: | `-n, --limit ` | `list` | Cap the rows (non-negative; `0` shows none). | | `--format ` | `export` | Output format. Default `md`. | | `-o, --out ` | `export` | Write to a file instead of stdout. | -| `--dry-run` | `batch`, `prune` | Show what would go; submit/delete nothing. | +| `--dry-run` | `research`, `batch`, `prune` | Show what would go — including the projected agent spend; submit/delete nothing. | | `--raw` | `get` | Dump the raw interaction JSON. | IDs are the first 8 characters of a UUID. Any unique prefix works, so `gemcatch get 8f3a` is fine. @@ -153,6 +155,46 @@ $ id=$(gemcatch research "..." --json | jq -r .id) $ gemcatch watch "$id" --json | jq -r .result ``` +## Research agents + +The [Gemini Deep Research agents](https://ai.google.dev/gemini-api/docs/deep-research) are reachable only through the Interactions API, and the docs are explicit: *"You must use background execution (set `background=true`) to run the agent asynchronously and poll for results or stream updates."* That is precisely the half of the job `gemcatch` already does — it always sets `background: true`, owns the polling, and its daemon collects results before the free tier drops interactions after **1 day** (paid tier: 55 days). A Deep Research run takes minutes and you were never going to sit there holding the connection; submit it, and let the daemon catch it. + +```console +$ gemcatch research "map the EU AI Act high-risk obligations against the UK approach" --agent deep-research +Agent deep-research-preview-04-2026 — estimated $1.00–$3.00 for this task (preview rates, subject to change). +Submit? [y/N] y +Task 8f3a1c04 submitted. Run: gemcatch get 8f3a1c04 when ready. +``` + +`--agent` takes an alias or a raw agent id: + +| You type | Sent to the API | +| --- | --- | +| `deep-research` | `deep-research-preview-04-2026` | +| `deep-research-max` | `deep-research-max-preview-04-2026` | +| anything else | passed through unchanged (future agent ids work without a gemcatch release; a bad id fails fast with the API's own 4xx) | + +An agent is sent **instead of** a model — the agent picks its own models — so `--model` and `--agent` together is an error, and nothing is submitted. + +**These agents cost real money, per task.** The docs put Deep Research at **$1.00–$3.00 per task** and Deep Research Max at **$3.00–$7.00 per task** — with their own hedge attached: *"These figures are estimates based on preview rates and are subject to change."* Because `gemcatch batch` fires a whole file at once, a 20-line file against `deep-research-max` is a **$60–$140 command**, so every agent submission shows its band and asks first. In a script (stdin not a TTY) you must pass `--yes`; `--dry-run` prints the full projected spend and submits nothing: + +```console +$ gemcatch batch questions.txt --agent deep-research-max --dry-run +20 prompts × deep-research-max-preview-04-2026 — estimated $60.00–$140.00 total. Nothing submitted (--dry-run). +``` + +The report lands like any other result — final answer only, none of the agent's interim plan — and its **citations** come with it. The docs tell you to review them to verify the sources, so `gemcatch get` prints them under the report as a `Sources:` list, `--json` carries them as an array, and they live in the store alongside the result. + +An agent run can also come back `incomplete` — that is what a `max_total_tokens` budget cap produces when the run "safely pauses" — which `gemcatch` treats as terminal, exactly like the API does: the daemon retires it and moves on. + +The agent recipe, end to end: + +```bash +$ gemcatch batch questions.txt --agent deep-research --yes # bands shown, N × total quoted +$ gemcatch daemon --exit-when-idle # catch reports before the 1-day expiry +$ gemcatch export --tag batch-1a2b3c -o reports.md # every report, with its sources +``` + ## How it works Tasks live in SQLite at `~/.gemcatch/tasks.db` (override with `GEMCATCH_HOME`): @@ -160,7 +202,7 @@ Tasks live in SQLite at `~/.gemcatch/tasks.db` (override with `GEMCATCH_HOME`): ```sql CREATE TABLE tasks (id TEXT PRIMARY KEY, prompt TEXT, interaction_id TEXT, status TEXT DEFAULT 'pending', result TEXT, created_at INTEGER); --- plus model, system_instruction, tag, error, usage, updated_at +-- plus model, system_instruction, tag, error, usage, updated_at, agent, citations ``` `research` calls `interactions.create({model, input, background: true})` via [`@google/genai`](https://www.npmjs.com/package/@google/genai) and keeps the returned `id`. The polling commands call `interactions.get(id)` and write the status back. Once a task completes, the text is cached in the `result` column — `gemcatch get` then answers from disk without touching the network. @@ -208,7 +250,7 @@ Transient failures are retried with exponential backoff and full jitter, honouri | --- | --- | | `GEMINI_API_KEY` | Your API key. `GOOGLE_API_KEY` also works. | | `GEMCATCH_HOME` | Where `tasks.db` lives. Default `~/.gemcatch`. | -| `GEMCATCH_MODEL` | Default model. Default `gemini-3.1-flash-lite`. | +| `GEMCATCH_MODEL` | Default model. Default `gemini-3.5-flash-lite`. | | `GEMCATCH_POLL_MS` | `watch` poll interval in ms. Default `10000`. | | `GEMCATCH_DAEMON_S` | `daemon` interval in seconds. Default `300`. | | `GEMCATCH_RPM` | Requests/minute ceiling. Default `15` (the free tier). `0` disables pacing. | diff --git a/db.js b/db.js index 77b8c6b..2c02600 100644 --- a/db.js +++ b/db.js @@ -26,6 +26,11 @@ const MIGRATIONS = [ ['error', 'TEXT'], ['usage', 'TEXT'], ['updated_at', 'INTEGER'], + // 0.4.0: agent runs. `agent` is the resolved agent id the task was submitted + // with (NULL for model runs, including every pre-0.4.0 row); `citations` is + // the JSON array of sources an agent run returned alongside its report. + ['agent', 'TEXT'], + ['citations', 'TEXT'], ]; let _db = null; @@ -59,8 +64,8 @@ function createTask(fields) { const now = Date.now(); db() .prepare( - 'INSERT INTO tasks (id, prompt, status, created_at, updated_at, model, system_instruction, tag) ' + - 'VALUES (@id, @prompt, @status, @now, @now, @model, @system_instruction, @tag)' + 'INSERT INTO tasks (id, prompt, status, created_at, updated_at, model, system_instruction, tag, agent) ' + + 'VALUES (@id, @prompt, @status, @now, @now, @model, @system_instruction, @tag, @agent)' ) .run({ id, @@ -70,6 +75,7 @@ function createTask(fields) { model: t.model || null, system_instruction: t.systemInstruction || null, tag: t.tag || null, + agent: t.agent || null, }); return id; } @@ -101,7 +107,7 @@ function setStatus(id, status, extra) { const e = extra || {}; const sets = ['status = @status', 'updated_at = @now']; const params = { id, status, now: Date.now() }; - for (const key of ['result', 'error', 'usage']) { + for (const key of ['result', 'error', 'usage', 'citations']) { if (e[key] !== undefined) { sets.push(`${key} = @${key}`); params[key] = e[key]; @@ -167,6 +173,14 @@ function counts() { return db().prepare('SELECT status, COUNT(*) AS n FROM tasks GROUP BY status').all(); } +// Per-agent totals for `stats`. Model runs (agent IS NULL) are not a row here; +// they are already accounted for in counts(). +function agentCounts() { + return db() + .prepare('SELECT agent, COUNT(*) AS n FROM tasks WHERE agent IS NOT NULL GROUP BY agent') + .all(); +} + function close() { if (_db) _db.close(); _db = null; @@ -185,5 +199,6 @@ module.exports = { removeMany, prunableTasks, counts, + agentCounts, close, }; diff --git a/gemini.js b/gemini.js index a57e349..10568a6 100644 --- a/gemini.js +++ b/gemini.js @@ -3,7 +3,41 @@ const { isDone, isSuccess } = require('./status'); // Free of charge on the Gemini free tier; override per-call with --model. -const DEFAULT_MODEL = process.env.GEMCATCH_MODEL || 'gemini-3.1-flash-lite'; +// gemini-3.5-flash-lite went GA on 2026-07-21 (it replaced 3.1 as the +// low-latency free-tier workhorse in the same release that deprecated the +// sampling parameters). +const DEFAULT_MODEL = process.env.GEMCATCH_MODEL || 'gemini-3.5-flash-lite'; + +// --- agents --------------------------------------------------------------- + +// The Deep Research agents are reachable ONLY through the Interactions API, +// and only with background execution -- which gemcatch always sets. An agent +// is sent as `agent` on create, INSTEAD of `model`: the two are mutually +// exclusive, and the CLI rejects the combination before anything is written. +// +// This table is the ONE place the full preview ids live. They are preview ids +// and will be superseded; call sites must resolve through here (or pass an +// unknown id straight through, so a future agent works without a release). +const AGENT_ALIASES = Object.freeze({ + 'deep-research': 'deep-research-preview-04-2026', + 'deep-research-max': 'deep-research-max-preview-04-2026', +}); + +// Documented per-task price bands, in dollars, keyed by the RESOLVED id. +// The docs' own hedge applies -- "These figures are estimates based on +// preview rates and are subject to change" -- so the spend guard quotes +// them as estimates, never as authoritative. +const AGENT_PRICE_BANDS = Object.freeze({ + 'deep-research-preview-04-2026': Object.freeze([1, 3]), + 'deep-research-max-preview-04-2026': Object.freeze([3, 7]), +}); + +// A known alias resolves to its full preview id; anything else passes through +// unchanged so a new or newer agent id works without a gemcatch release (a +// genuinely bad id fails fast: the API 4xxes, and a 4xx never retries). +function resolveAgent(id) { + return AGENT_ALIASES[id] || id; +} // Overridable for tests and for routing via a proxy/gateway. const REST_BASE = @@ -163,7 +197,11 @@ function collectText(node, acc) { return acc; } if (typeof node.text === 'string' && node.text.trim()) acc.push(node.text); - for (const v of Object.values(node)) { + for (const [k, v] of Object.entries(node)) { + // Citations are sources *about* the answer, not answer text: an agent step + // carries them alongside its content, and a citation's own title/snippet + // must not be concatenated into the result. They are collected separately. + if (k === 'citations') continue; if (v && typeof v === 'object') collectText(v, acc); } return acc; @@ -173,21 +211,65 @@ function collectText(node, acc) { // internal reasoning with the actual answer, each tagged by `type`: // [ {type:'user_input', ...}, {type:'thought', ...}, {type:'model_output', ...} ] // Collecting text indiscriminately prepends the prompt (and any reasoning) to -// the result, so those step types are skipped. Anything else -- model_output, -// an untyped step, a future answer-bearing type -- still contributes, so a -// renamed step never silently blanks the result. +// the result, so those step types are skipped. +// +// Both kinds of run put the deliverable in the FINAL answer-bearing step. A +// model run ends [user_input, thought, model_output]; an agent run's steps +// additionally interleave its plan, searches and interim drafts, and the docs +// place the finished report at `interaction.steps[-1].content[0].text`. So one +// rule serves both, with no special-casing on the agent id: take the last step +// that is not user_input/thought. If that step somehow carries no text -- an +// unexpected shape, a renamed type -- fall back to collecting across every +// answer-bearing step, so the failure mode is "too much text", never a +// silently blank result. const NON_ANSWER_STEP = new Set(['user_input', 'thought']); function textFromSteps(steps) { if (!Array.isArray(steps)) return ''; + const candidates = steps.filter((s) => !(s && NON_ANSWER_STEP.has(s.type))); + if (!candidates.length) return ''; + const last = collectText(candidates[candidates.length - 1], []).join('\n').trim(); + if (last) return last; const acc = []; - for (const step of steps) { - if (step && NON_ANSWER_STEP.has(step.type)) continue; - collectText(step, acc); - } + for (const step of candidates) collectText(step, acc); return acc.join('\n').trim(); } +// Agent runs carry citations -- the docs explicitly tell users to review them +// to verify the sources -- so they are gathered rather than discarded. The +// walk is shape-agnostic (any `citations` array anywhere in the interaction), +// because the docs do not pin down where they attach; duplicates are dropped. +function collectCitations(node, acc) { + if (!node || typeof node !== 'object') return acc; + if (Array.isArray(node)) { + for (const n of node) collectCitations(n, acc); + return acc; + } + for (const [k, v] of Object.entries(node)) { + if (k === 'citations' && Array.isArray(v)) { + for (const c of v) if (c && typeof c === 'object') acc.push(c); + continue; + } + if (v && typeof v === 'object') collectCitations(v, acc); + } + return acc; +} + +function citationsOf(interaction) { + const all = collectCitations(interaction, []); + if (!all.length) return null; + const seen = new Set(); + const out = []; + for (const c of all) { + const key = JSON.stringify(c); + if (!seen.has(key)) { + seen.add(key); + out.push(c); + } + } + return out; +} + function textOf(interaction) { if (interaction && typeof interaction.output_text === 'string' && interaction.output_text) { return interaction.output_text; @@ -200,6 +282,7 @@ function shape(r) { interactionId: r.id, status: r.status, text: textOf(r), + citations: citationsOf(r), usage: r.usage || null, raw: r, }; @@ -269,7 +352,13 @@ function restHeaders() { async function submit(prompt, opts) { const o = opts || {}; - const body = { model: o.model || DEFAULT_MODEL, input: prompt, background: true }; + // `agent` and `model` are mutually exclusive on create: an agent run is sent + // with `agent` INSTEAD of `model` (the agent picks its own models). `input` + // stays a plain string and `background` stays true either way -- agents + // *require* background execution, which gemcatch has always set. + const body = o.agent + ? { agent: o.agent, input: prompt, background: true } + : { model: o.model || DEFAULT_MODEL, input: prompt, background: true }; if (o.systemInstruction) body.system_instruction = o.systemInstruction; const r = await call(() => { const api = sdkInteractions(); @@ -323,6 +412,9 @@ module.exports = { REST_BASE, RPM, MAX_RETRIES, + AGENT_ALIASES, + AGENT_PRICE_BANDS, + resolveAgent, submit, poll, cancel, @@ -330,6 +422,7 @@ module.exports = { apiKey, textOf, collectText, + citationsOf, // Exported for the suite: the retry policy is behaviour worth pinning. shouldRetry, // Re-exported so callers need only one require. diff --git a/index.js b/index.js index fabe923..1ce1a5f 100644 --- a/index.js +++ b/index.js @@ -89,6 +89,104 @@ function needTask(id) { return task; } +// Citations ride along with an agent's report -- the docs tell users to review +// them to verify the sources, so they are printed under the result rather than +// left in the database. A run without citations prints exactly as before. +function withSources(text, citations) { + const body = text || '(empty response)'; + if (!Array.isArray(citations) || !citations.length) return body; + const lines = citations.map((c, i) => { + const title = (c && (c.title || c.text)) || ''; + const url = (c && (c.url || c.uri)) || ''; + return ` [${i + 1}] ${[title, url].filter(Boolean).join(' — ') || JSON.stringify(c)}`; + }); + return `${body}\n\nSources:\n${lines.join('\n')}`; +} + +// The citations column holds JSON (or NULL). Parsed defensively: a corrupt row +// degrades to "no sources", never a crash in the middle of printing a result. +function parseCitations(raw) { + if (!raw) return null; + try { + const v = JSON.parse(raw); + return Array.isArray(v) && v.length ? v : null; + } catch (_) { + return null; + } +} + +// --- spend guard ---------------------------------------------------------- + +// Deep Research agents are billed PER TASK, not per token -- the docs put +// Deep Research at $1.00-$3.00 and Deep Research Max at $3.00-$7.00 -- and +// gemcatch's whole ergonomic is firing a file of prompts at once, which turns +// one careless `batch --agent` into a three-figure command. So no agent +// submission happens without the cost being shown and confirmed: interactively +// on a TTY, via --yes otherwise, and --dry-run previews without submitting. +// The bands are quoted with the docs' own hedge ("estimates based on preview +// rates and subject to change"), never as authoritative. + +function bandText(agentId, count) { + const band = gemini.AGENT_PRICE_BANDS[agentId]; + if (!band) return 'no published price band for this agent'; + const money = (n) => `$${(n * count).toFixed(2)}`; + return count > 1 + ? `estimated ${money(band[0])}–${money(band[1])} total` + : `estimated ${money(band[0])}–${money(band[1])} for this task`; +} + +function spendLine(agentId, count) { + const head = count > 1 ? `${count} prompts × ${agentId}` : `Agent ${agentId}`; + return `${head} — ${bandText(agentId, count)}`; +} + +function askYesNo(question) { + const readline = require('readline'); + const rl = readline.createInterface({ input: process.stdin, output: process.stderr }); + return new Promise((resolve) => { + rl.question(question, (answer) => { + rl.close(); + resolve(/^y(es)?$/i.test((answer || '').trim())); + }); + }); +} + +// Returns only when the submission is confirmed; otherwise it exits (declined) +// or throws (no way to ask). Runs BEFORE any row is written, so a declined or +// refused submission leaves the tasks table untouched. +async function confirmSpend(agentId, count, opts) { + console.error(`${spendLine(agentId, count)} (preview rates, subject to change).`); + if (opts.yes) return; + // GEMCATCH_ASSUME_TTY lets the offline suite drive the interactive branch + // through a pipe; real non-TTY callers (cron, CI, scripts) must say --yes. + const interactive = process.stdin.isTTY || process.env.GEMCATCH_ASSUME_TTY === '1'; + if (!interactive) { + throw new Error( + 'stdin is not a TTY, so this agent submission cannot be confirmed interactively.\n' + + ' Pass --yes to confirm the cost above, or --dry-run to preview without submitting.' + ); + } + if (!(await askYesNo('Submit? [y/N] '))) { + console.error('Nothing submitted.'); + process.exit(1); + } +} + +// Shared by research and batch: resolve the agent alias and reject the +// ambiguous combination before anything is stored or sent. `--model` counts +// only when the user actually typed it -- commander fills in the default +// otherwise, and the default must not poison every agent run. +function resolveAgentOpts(opts, cmd) { + if (!opts.agent) return null; + if (cmd.getOptionValueSource('model') === 'cli') { + throw new Error( + '--model and --agent are mutually exclusive: an agent run is submitted with `agent` ' + + 'instead of `model`, and the agent picks its own models. Drop one of the two.' + ); + } + return gemini.resolveAgent(opts.agent); +} + // --- input ---------------------------------------------------------------- function readStdin() { @@ -139,8 +237,12 @@ async function refresh(task) { } const extra = {}; if (isDone(r.status)) { - if (isSuccess(r.status)) extra.result = r.text; - else if (r.text) extra.error = r.text; + if (isSuccess(r.status)) { + extra.result = r.text; + // Agent runs return citations with the report; the docs tell users to + // review them to verify the sources, so they are persisted, not dropped. + if (r.citations && r.citations.length) extra.citations = JSON.stringify(r.citations); + } else if (r.text) extra.error = r.text; } if (r.usage) extra.usage = JSON.stringify(r.usage); store.setStatus(task.id, r.status, extra); @@ -177,22 +279,35 @@ program .argument('[prompt]', 'what you want researched; "-" reads stdin') .option('-f, --file ', 'read the prompt from a file') .option('-m, --model ', 'model to use', gemini.DEFAULT_MODEL) + .option('-a, --agent ', 'submit to a research agent instead of a model (e.g. deep-research)') .option('-s, --system ', 'system instruction') .option('-t, --tag ', 'label for filtering with `gemcatch list --tag`') .option('-w, --watch', 'wait for the result instead of exiting') + .option('--yes', 'confirm the agent cost without asking (required when stdin is not a TTY)') + .option('--dry-run', 'show what would be submitted (and what it would cost); submit nothing') .option('--json', 'machine-readable output') .description('submit a background task and exit immediately') - .action(async (promptArg, opts) => { + .action(async (promptArg, opts, cmd) => { let id; try { + const agent = resolveAgentOpts(opts, cmd); const prompt = await resolvePrompt(promptArg, opts); + if (opts.dryRun) { + emit(opts.json, { dry_run: true, agent: agent || null, model: agent ? null : opts.model, prompt }, () => { + if (agent) console.log(`${spendLine(agent, 1)}. Nothing submitted (--dry-run).`); + else console.log(`Would submit to ${opts.model}: ${snippet(prompt)}. Nothing submitted (--dry-run).`); + }); + return; + } + if (agent) await confirmSpend(agent, 1, opts); id = store.createTask({ prompt, - model: opts.model, + model: agent ? null : opts.model, + agent, systemInstruction: opts.system, tag: opts.tag, }); - const r = await gemini.submit(prompt, { model: opts.model, systemInstruction: opts.system }); + const r = await gemini.submit(prompt, { model: opts.model, agent, systemInstruction: opts.system }); store.setInteraction(id, r.interactionId, r.status); if (opts.watch) { // Under --watch the submit line is progress, not the answer, so it @@ -296,15 +411,18 @@ program .command('batch') .argument('', 'prompts file — one per line, or "-" to read stdin') .option('-m, --model ', 'model to use', gemini.DEFAULT_MODEL) + .option('-a, --agent ', 'submit every prompt to a research agent instead of a model') .option('-s, --system ', 'system instruction') .option('-t, --tag ', 'tag the whole batch (default: batch-)') .option('--separator ', 'split the file on this delimiter line for multi-line prompts') .option('-w, --watch', 'submit all, then poll until the whole batch finishes') + .option('--yes', 'confirm the agent cost without asking (required when stdin is not a TTY)') .option('--dry-run', 'parse and list what would be submitted; submit nothing') .option('--json', 'machine-readable output') .description('submit many background tasks from a file, tagged as one batch') - .action(async (file, opts) => { + .action(async (file, opts, cmd) => { try { + const agent = resolveAgentOpts(opts, cmd); const text = file === '-' ? await readStdin() : fs.readFileSync(file, 'utf8'); const { prompts, skipped } = parsePrompts(text, opts.separator); if (!prompts.length) throw new Error(`no prompts found in ${file === '-' ? 'stdin' : file}`); @@ -317,19 +435,28 @@ program const tag = opts.tag || `batch-${crypto.randomUUID().slice(0, 6)}`; if (opts.dryRun) { - emit(opts.json, { tag, dry_run: true, prompts }, () => { - console.log(`Batch ${tag}: ${prompts.length} prompt(s) would be submitted:`); - for (const p of prompts) console.log(` ${snippet(p)}`); + emit(opts.json, { tag, dry_run: true, agent: agent || null, prompts }, () => { + if (agent) { + // The whole point of the guard: N × the per-task band, up front. + console.log(`${spendLine(agent, prompts.length)}. Nothing submitted (--dry-run).`); + } else { + console.log(`Batch ${tag}: ${prompts.length} prompt(s) would be submitted:`); + for (const p of prompts) console.log(` ${snippet(p)}`); + } }); return; } + // An agent batch multiplies a per-task dollar band by the whole file, so + // it is confirmed as one total before a single row is written. + if (agent) await confirmSpend(agent, prompts.length, opts); + // One failed submit must not sink the batch: mark that task failed and // keep going. mapLimit preserves input order, so the report is stable. const results = await mapLimit(prompts, 4, async (prompt) => { - const id = store.createTask({ prompt, model: opts.model, systemInstruction: opts.system, tag }); + const id = store.createTask({ prompt, model: agent ? null : opts.model, agent, systemInstruction: opts.system, tag }); try { - const r = await gemini.submit(prompt, { model: opts.model, systemInstruction: opts.system }); + const r = await gemini.submit(prompt, { model: opts.model, agent, systemInstruction: opts.system }); store.setInteraction(id, r.interactionId, r.status); return { id, interaction_id: r.interactionId, status: r.status, prompt }; } catch (err) { @@ -402,16 +529,17 @@ program // text stores `''`, which is exactly the case the cache must still serve // -- re-polling it would 404 after 24h, the very thing we cache to avoid. if (isSuccess(task.status) && task.result != null && !opts.raw) { - emit(opts.json, { id: task.id, status: task.status, result: task.result }, () => - console.log(task.result || '(empty response)') + const cits = parseCitations(task.citations); + emit(opts.json, { id: task.id, status: task.status, result: task.result, citations: cits }, () => + console.log(withSources(task.result, cits)) ); return; } const r = await refresh(task); if (opts.raw) return console.log(JSON.stringify(r.raw, null, 2)); if (isSuccess(r.status)) { - emit(opts.json, { id: task.id, status: r.status, result: r.text }, () => - console.log(r.text || '(empty response)') + emit(opts.json, { id: task.id, status: r.status, result: r.text, citations: r.citations || null }, () => + console.log(withSources(r.text, r.citations)) ); } else if (isDone(r.status)) { emit(opts.json, { id: task.id, status: r.status, error: r.text || null }, () => @@ -450,14 +578,21 @@ program console.log('No tasks yet. Submit one: gemcatch research "your question"'); return; } - console.log(dim('ID AGE STATUS PROMPT')); + // The AGENT column only appears when something in the listing used one, so + // a pure-model store keeps the compact four-column layout it always had. + // Agent ids are shown compact -- the "-preview-MM-YYYY" suffix is version + // noise in a table (the full id is in --json and in stats). + const showAgent = tasks.some((t) => t.agent); + const shortAgent = (a) => (a ? a.replace(/-preview-\d{2}-\d{4}$/, '') : '-'); + console.log(dim(`ID AGE STATUS ${showAgent ? 'AGENT ' : ''}PROMPT`)); for (const t of tasks) { const snip = snippet(t.prompt); const status = t.status || PENDING; // Pad before colouring: ANSI codes would break the column width. const pad = ' '.repeat(Math.max(0, 16 - status.length)); + const agentCol = showAgent ? `${shortAgent(t.agent).padEnd(18)} ` : ''; console.log( - `${t.id} ${age(t.created_at).padEnd(4)} ${colorStatus(status)}${pad} ${snip}` + `${t.id} ${age(t.created_at).padEnd(4)} ${colorStatus(status)}${pad} ${agentCol}${snip}` ); } }); @@ -726,8 +861,8 @@ async function watchTask(task, intervalMs, json) { last = r.status; } if (isSuccess(r.status)) { - emit(json, { id: task.id, status: r.status, result: r.text }, () => - console.log(r.text || '(empty response)') + emit(json, { id: task.id, status: r.status, result: r.text, citations: r.citations || null }, () => + console.log(withSources(r.text, r.citations)) ); return; } @@ -755,8 +890,9 @@ program // Serve a completed result from cache -- present, not merely truthy, so an // empty-text completion is served instead of re-polled (and lost at 24h). if (isSuccess(task.status) && task.result != null) { - emit(opts.json, { id: task.id, status: task.status, result: task.result }, () => - console.log(task.result || '(empty response)') + const cits = parseCitations(task.citations); + emit(opts.json, { id: task.id, status: task.status, result: task.result, citations: cits }, () => + console.log(withSources(task.result, cits)) ); return; } @@ -850,11 +986,16 @@ program .description('where the store lives and what is in it') .action((opts) => { const rows = store.counts(); + const agents = store.agentCounts(); const total = rows.reduce((n, r) => n + r.n, 0); - emit(opts.json, { db: store.DB_PATH, total, by_status: rows }, () => { + emit(opts.json, { db: store.DB_PATH, total, by_status: rows, by_agent: agents }, () => { console.log(`Store: ${store.DB_PATH}`); console.log(`Tasks: ${total}`); for (const r of rows) console.log(` ${colorStatus(r.status).padEnd(useColor ? 26 : 17)} ${r.n}`); + if (agents.length) { + console.log('Agent runs:'); + for (const a of agents) console.log(` ${a.agent.padEnd(34)} ${a.n}`); + } }); }); diff --git a/package.json b/package.json index eebef17..2a84a6a 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "gemcatch", - "version": "0.3.0", + "version": "0.4.0", "description": "Fire-and-forget CLI for Gemini's Interactions API background execution. Submit long-running research prompts, close your laptop, collect results later.", "main": "index.js", "bin": { @@ -29,6 +29,7 @@ "background", "async", "agents", + "deep-research", "cli", "research", "daemon", diff --git a/test-offline.js b/test-offline.js index 19f6071..507dfbf 100644 --- a/test-offline.js +++ b/test-offline.js @@ -21,13 +21,26 @@ const { spawn } = require('child_process'); const Database = require('better-sqlite3'); const ANSWER = 'This week in AI: the Interactions API shipped background execution.'; +// An agent's report lives in the FINAL step; the interim steps hold the plan +// and drafts and must never leak into the result. +const AGENT_ANSWER = + 'Deep Research report: the EU AI Act phases in high-risk obligations from August 2026, ' + + 'while the UK relies on regulator-led guidance.'; +const AGENT_CITATIONS = [ + { title: 'EU AI Act — EUR-Lex', url: 'https://eur-lex.europa.eu/eli/reg/2024/1689' }, + { title: 'UK AI regulation white paper', url: 'https://www.gov.uk/ai-regulation-pro-innovation' }, +]; +// The mock accepts exactly the documented preview ids; anything else 4xxes, +// like the real API would for a bad agent id. +const KNOWN_AGENTS = new Set(['deep-research-preview-04-2026', 'deep-research-max-preview-04-2026']); const HOME = fs.mkdtempSync(path.join(os.tmpdir(), 'gemcatch-test-')); let seq = 0; let getHits = 0; let keyRejects = 0; let flaky503s = 0; -const interactions = new Map(); // id -> {status, pollsLeft, text, model, system, deleted} +let agentRejects = 0; +const interactions = new Map(); // id -> {status, pollsLeft, text, model, system, agent, deleted} // --- mock Interactions API ------------------------------------------------ @@ -52,6 +65,15 @@ const server = http.createServer((req, res) => { const body = JSON.parse(raw); assert.strictEqual(body.background, true, 'background must be true'); assert.strictEqual(typeof body.input, 'string', 'input must be a plain string'); + // `agent` replaces `model` on create; the two are mutually exclusive and + // exactly one must be present. The CLI enforces this before submitting, + // so the mock asserting it catches any regression that slips one through. + assert(!(body.agent && body.model), 'agent and model are mutually exclusive'); + assert(body.agent || body.model, 'one of agent or model is required'); + if (body.agent && !KNOWN_AGENTS.has(body.agent)) { + agentRejects += 1; + return send(400, { error: { code: 400, message: `Unknown agent id: ${body.agent}.` } }); + } const id = `int_${++seq}`; interactions.set(id, { status: 'in_progress', @@ -69,6 +91,10 @@ const server = http.createServer((req, res) => { // drive the watch/daemon consecutive-failure safety bound. hardFailLeft: /HARDFAIL/.test(body.input) ? Infinity : /WATCHWEDGE/.test(body.input) ? 4 : 0, model: body.model, + agent: body.agent, + // BUDGETPAUSE mimics a max_total_tokens cap being hit: the agent run + // "safely pauses" and the interaction comes back status: incomplete. + budgetPause: /BUDGETPAUSE/.test(body.input), system: body.system_instruction, }); send(200, { id, status: 'in_progress' }); @@ -119,6 +145,37 @@ const server = http.createServer((req, res) => { } if (it.status === 'cancelled') return send(200, { id: idPart, status: 'cancelled' }); if (it.fails) return send(200, { id: idPart, status: 'failed', usage: { total_tokens: 3 } }); + if (it.agent) { + // A max_total_tokens budget pause: the run stops part-way and the + // interaction returns status incomplete -- terminal, with no report. + if (it.budgetPause) { + return send(200, { + id: idPart, + agent: it.agent, + status: 'incomplete', + usage: { total_tokens: 500 }, + steps: [ + { type: 'user_input', content: [{ type: 'text', text: it.prompt }] }, + { type: 'thought', signature: 'redacted' }, + ], + }); + } + // An agent run is multi-step: plan and interim drafts first, the actual + // report in the FINAL step (docs: interaction.steps[-1].content[0].text), + // with citations attached. Only that final step's text is the answer. + return send(200, { + id: idPart, + agent: it.agent, + status: 'completed', + usage: { total_tokens: 1234 }, + steps: [ + { type: 'user_input', content: [{ type: 'text', text: it.prompt }] }, + { type: 'thought', signature: 'redacted' }, + { type: 'model_output', content: [{ type: 'text', text: 'Interim: research plan drafted, 12 sources fetched.' }] }, + { type: 'model_output', content: [{ type: 'text', text: AGENT_ANSWER }], citations: AGENT_CITATIONS }, + ], + }); + } // No output_text on REST: text must be recovered from steps. The real API // interleaves the echoed prompt and the model's reasoning with the answer, // so the mock does too -- only the model_output text may come back. @@ -209,6 +266,31 @@ function cliTimeout(args, extra, ms) { } const out = async (args, extra) => (await cli(args, extra)).stdout; + +// Open-query-close helpers: on Windows a leaked better-sqlite3 handle keeps +// tasks.db locked and the suite's final rmSync dies with EBUSY. +function qget(home, sql, ...params) { + const d = new Database(path.join(home, 'tasks.db'), { readonly: true }); + try { + return d.prepare(sql).get(...params); + } finally { + d.close(); + } +} +function qall(home, sql, ...params) { + const d = new Database(path.join(home, 'tasks.db'), { readonly: true }); + try { + return d.prepare(sql).all(...params); + } finally { + d.close(); + } +} +// Row count that treats "no db yet" as zero -- a refused submission may exit +// before the store is even created, and both outcomes are "nothing written". +function taskCount(home) { + if (!fs.existsSync(path.join(home, 'tasks.db'))) return 0; + return qget(home, 'SELECT COUNT(*) AS n FROM tasks').n; +} const idOf = (text) => (text.match(/^Task (\w+) submitted\./m) || [])[1]; const ok = (name) => console.log(` ok ${name}`); @@ -244,6 +326,35 @@ async function submit(prompt, args, extra) { ); ok('textOf prefers output_text, falls back to steps'); + // Agent runs: the report is the FINAL answer-bearing step; the interim + // drafts must not be concatenated in, and citations are metadata, never text. + const agentShaped = { + agent: 'deep-research-preview-04-2026', + steps: [ + { type: 'user_input', content: [{ type: 'text', text: 'the prompt' }] }, + { type: 'thought', signature: 'redacted' }, + { type: 'model_output', content: [{ type: 'text', text: 'interim draft' }] }, + { + type: 'model_output', + content: [{ type: 'text', text: 'the final report' }], + citations: [{ title: 'A Source', url: 'https://example.com/a' }], + }, + ], + }; + assert.strictEqual(gemini.textOf(agentShaped), 'the final report'); + assert.deepStrictEqual(gemini.citationsOf(agentShaped), [{ title: 'A Source', url: 'https://example.com/a' }]); + assert.strictEqual(gemini.citationsOf({ steps: [] }), null, 'a model run has no citations'); + ok('textOf takes the final step of an agent run; citationsOf gathers sources, textOf excludes them'); + + // ONE alias table: known aliases resolve to the full preview ids, anything + // else passes through untouched so future agent ids need no gemcatch release. + assert.strictEqual(gemini.resolveAgent('deep-research'), 'deep-research-preview-04-2026'); + assert.strictEqual(gemini.resolveAgent('deep-research-max'), 'deep-research-max-preview-04-2026'); + assert.strictEqual(gemini.resolveAgent('some-future-agent-01-2027'), 'some-future-agent-01-2027'); + assert(gemini.AGENT_PRICE_BANDS['deep-research-preview-04-2026'], 'the standard band exists'); + assert(gemini.AGENT_PRICE_BANDS['deep-research-max-preview-04-2026'], 'the max band exists'); + ok('resolveAgent maps aliases through one table and passes unknown ids through'); + // A 4xx is a bad request and will fail identically forever; retrying it just // wastes the user's rate limit. Everything transient gets another go. for (const s of [408, 429, 500, 502, 503, 504]) { @@ -276,11 +387,33 @@ async function submit(prompt, args, extra) { const migrated = (await out(['list'], { env: { GEMCATCH_HOME: legacy } })); assert(migrated.includes('old00001'), `legacy row should survive migration: ${migrated}`); const cols = new Database(path.join(legacy, 'tasks.db')).prepare('PRAGMA table_info(tasks)').all().map((c) => c.name); - for (const c of ['model', 'tag', 'usage', 'updated_at', 'error', 'system_instruction']) { + for (const c of ['model', 'tag', 'usage', 'updated_at', 'error', 'system_instruction', 'agent', 'citations']) { assert(cols.includes(c), `migration should add ${c}`); } ok('a v1 tasks.db migrates in place without losing rows'); + // ---- v0.3.0 -> 0.4.0 migration ---- + // A store written by 0.3.0 (all pre-agent columns, no agent/citations) must + // upgrade in place: every row preserved, agent reported as NULL for them. + const v030 = path.join(HOME, 'v030'); + fs.mkdirSync(v030, { recursive: true }); + const v030Db = new Database(path.join(v030, 'tasks.db')); + v030Db.exec( + "CREATE TABLE tasks (id TEXT PRIMARY KEY, prompt TEXT, interaction_id TEXT, status TEXT DEFAULT 'pending', " + + 'result TEXT, created_at INTEGER, model TEXT, system_instruction TEXT, tag TEXT, error TEXT, usage TEXT, updated_at INTEGER)' + ); + const insV030 = v030Db.prepare( + 'INSERT INTO tasks (id, prompt, status, result, created_at, model) VALUES (?,?,?,?,?,?)' + ); + insV030.run('pre04001', 'a 0.3.0 row', 'completed', 'old answer', 100, 'gemini-3.1-flash-lite'); + insV030.run('pre04002', 'another 0.3.0 row', 'in_progress', null, 200, 'gemini-3.1-flash-lite'); + v030Db.close(); + const v030Rows = JSON.parse(await out(['list', '--json'], { env: { GEMCATCH_HOME: v030 } })); + assert.strictEqual(v030Rows.length, 2, 'every 0.3.0 row survives the 0.4.0 migration'); + for (const r of v030Rows) assert.strictEqual(r.agent, null, `a pre-agent row reports agent as NULL: ${JSON.stringify(r)}`); + assert.strictEqual(v030Rows.find((r) => r.id === 'pre04001').result, 'old answer', 'results are untouched'); + ok('a v0.3.0 tasks.db migrates to 0.4.0 preserving every row, agent NULL for all of them'); + // ---- research ---- const t0 = Date.now(); const first = await out(['research', 'summarize this week in AI', '--tag', 'ai']); @@ -301,8 +434,8 @@ async function submit(prompt, args, extra) { // ---- default model ---- const dflt = JSON.parse(await out(['research', 'default model', '--json'])); - assert.strictEqual(interactions.get(dflt.interaction_id).model, 'gemini-3.1-flash-lite', 'default model'); - ok('default model is gemini-3.1-flash-lite'); + assert.strictEqual(interactions.get(dflt.interaction_id).model, 'gemini-3.5-flash-lite', 'default model'); + ok('default model is gemini-3.5-flash-lite (GA since 2026-07-21)'); // ---- stdin + --file ---- const viaStdin = idOf(await out(['research', '-'], { stdin: 'prompt from stdin' })); @@ -713,6 +846,157 @@ async function submit(prompt, args, extra) { ); ok('digest feeds a tag\'s completed results through one Gemini call into a single summary'); + // ======================================================================== + // Research agents (0.4.0) + // ======================================================================== + + // ---- research --agent: alias resolves, agent (not model) reaches the API, + // the row records the full preview id, and the report comes from the final + // step with its citations persisted. ---- + const agEnv = { GEMCATCH_HOME: path.join(HOME, 'agent') }; + const agRun = await cli(['research', 'map the EU AI Act against the UK approach', '--agent', 'deep-research', '--yes'], { env: agEnv }); + const agId = idOf(agRun.stdout); + assert(agId, `agent research should submit: ${agRun.stdout}`); + assert( + /Agent deep-research-preview-04-2026 — estimated \$1\.00–\$3\.00 for this task \(preview rates, subject to change\)\./.test(agRun.stderr), + `the spend band must be shown even under --yes: ${agRun.stderr}` + ); + const agRow = qget(agEnv.GEMCATCH_HOME, 'SELECT agent, model, interaction_id FROM tasks WHERE id = ?', agId); + assert.strictEqual(agRow.agent, 'deep-research-preview-04-2026', 'the row stores the RESOLVED agent id'); + assert.strictEqual(agRow.model, null, 'an agent run stores no model'); + assert.strictEqual(interactions.get(agRow.interaction_id).agent, 'deep-research-preview-04-2026', 'the API got `agent`'); + assert.strictEqual(interactions.get(agRow.interaction_id).model, undefined, 'the API must NOT get `model`'); + await out(['get', agId], { env: agEnv }); // first poll: still in_progress + const agGet = await out(['get', agId], { env: agEnv }); + assert(agGet.includes(AGENT_ANSWER), `get should print the final-step report: ${agGet}`); + assert(!agGet.includes('Interim'), 'interim agent steps must not leak into the result'); + assert(agGet.includes('Sources:'), 'an agent result lists its sources'); + for (const c of AGENT_CITATIONS) assert(agGet.includes(c.url), `each citation url is printed: ${c.url}`); + const agCits = qget(agEnv.GEMCATCH_HOME, 'SELECT citations FROM tasks WHERE id = ?', agId); + assert.deepStrictEqual(JSON.parse(agCits.citations), AGENT_CITATIONS, 'citations are persisted as JSON'); + const agJson = JSON.parse(await out(['get', agId, '--json'], { env: agEnv })); + assert.deepStrictEqual(agJson.citations, AGENT_CITATIONS, 'get --json carries the citations from cache'); + ok('research --agent resolves the alias, sends agent instead of model, takes the final step, keeps citations'); + + // ---- list shows the agent; stats counts agent runs ---- + const agList = await out(['list'], { env: agEnv }); + assert(/AGENT/.test(agList), `list grows an AGENT column when an agent run exists: ${agList}`); + assert(/deep-research/.test(agList), 'the agent is shown against its task'); + const agStats = JSON.parse(await out(['stats', '--json'], { env: agEnv })); + assert.deepStrictEqual(agStats.by_agent, [{ agent: 'deep-research-preview-04-2026', n: 1 }], 'stats counts per agent'); + const agStatsHuman = await out(['stats'], { env: agEnv }); + assert(/Agent runs:/.test(agStatsHuman) && /deep-research-preview-04-2026/.test(agStatsHuman), `stats names the agent: ${agStatsHuman}`); + ok('list shows an AGENT column and stats tallies agent runs'); + + // ---- --model and --agent together: clean error, nothing submitted ---- + const mxEnv = { GEMCATCH_HOME: path.join(HOME, 'mutex') }; + const mx = await cli(['research', 'x', '--agent', 'deep-research', '--model', 'gemini-3.6-flash', '--yes'], { env: mxEnv }).catch((e) => e); + assert.strictEqual(mx.code, 1, '--model + --agent must be rejected'); + assert(/--model and --agent are mutually exclusive/.test(mx.stderr), `expected the mutual-exclusion message: ${mx.stderr}`); + const mxB = await cli(['batch', '-', '--agent', 'deep-research', '--model', 'gemini-3.6-flash', '--yes'], { env: mxEnv, stdin: 'q one\n' }).catch((e) => e); + assert.strictEqual(mxB.code, 1, 'batch rejects the combination too'); + assert(/mutually exclusive/.test(mxB.stderr), `batch should explain: ${mxB.stderr}`); + assert.strictEqual(taskCount(mxEnv.GEMCATCH_HOME), 0, 'a rejected flag combination writes no rows'); + ok('--model with --agent is a clean error and nothing is submitted'); + + // ---- unknown raw agent id: passed through, the 4xx surfaces immediately ---- + const unEnv = { GEMCATCH_HOME: path.join(HOME, 'unknown-agent') }; + const beforeAgentRejects = agentRejects; + const un = await cli(['research', 'probe', '--agent', 'brand-new-agent-01-2027', '--yes'], { env: unEnv }).catch((e) => e); + assert.strictEqual(un.code, 1, 'an unknown agent id fails'); + assert(/Unknown agent id: brand-new-agent-01-2027/.test(un.stderr), `the API's own 4xx message surfaces: ${un.stderr}`); + assert(/no published price band/.test(un.stderr), 'the guard is honest when it has no band for an id'); + assert.strictEqual(agentRejects - beforeAgentRejects, 1, 'a 4xx on submit is not retried'); + const unRow = qget(unEnv.GEMCATCH_HOME, 'SELECT status FROM tasks'); + assert.strictEqual(unRow.status, 'failed', 'the failed submit is recorded'); + ok('an unknown raw agent id passes through and the API 4xx surfaces immediately, unretried'); + + // ---- spend guard: declined -> zero rows, non-zero exit ---- + const dcEnv = { GEMCATCH_HOME: path.join(HOME, 'declined') }; + const dc = await cli(['research', 'expensive question', '--agent', 'deep-research-max'], + { env: Object.assign({}, dcEnv, { GEMCATCH_ASSUME_TTY: '1' }), stdin: 'n\n' }).catch((e) => e); + assert.strictEqual(dc.code, 1, 'declining the confirmation exits non-zero'); + assert(/\$3\.00–\$7\.00/.test(dc.stderr), `the max band is quoted before asking: ${dc.stderr}`); + assert(/Nothing submitted/.test(dc.stderr), `declining says so: ${dc.stderr}`); + assert.strictEqual(taskCount(dcEnv.GEMCATCH_HOME), 0, 'declining must leave the tasks table untouched'); + ok('declining the confirmation writes zero rows and exits non-zero'); + + // ---- spend guard: an interactive "y" submits ---- + const ycEnv = { GEMCATCH_HOME: path.join(HOME, 'confirmed') }; + const yc = await cli(['research', 'go ahead', '--agent', 'deep-research'], + { env: Object.assign({}, ycEnv, { GEMCATCH_ASSUME_TTY: '1' }), stdin: 'y\n' }); + assert(idOf(yc.stdout), `answering y submits: ${yc.stdout}`); + assert(/Submit\? \[y\/N\]/.test(yc.stderr), `the y/N question is asked on stderr: ${yc.stderr}`); + ok('answering y to the confirmation submits the task'); + + // ---- spend guard: non-TTY without --yes is refused ---- + const ntEnv = { GEMCATCH_HOME: path.join(HOME, 'notty') }; + const nt = await cli(['research', 'scripted', '--agent', 'deep-research'], { env: ntEnv }).catch((e) => e); + assert.strictEqual(nt.code, 1, 'non-TTY without --yes must refuse'); + assert(/stdin is not a TTY/.test(nt.stderr) && /--yes/.test(nt.stderr), `the refusal names the fix: ${nt.stderr}`); + assert.strictEqual(taskCount(ntEnv.GEMCATCH_HOME), 0, 'a refused submission writes no rows'); + ok('non-TTY without --yes is refused before anything is written'); + + // ---- research --agent --dry-run: band printed, nothing submitted ---- + const rdEnv = { GEMCATCH_HOME: path.join(HOME, 'research-dry') }; + const rd = await out(['research', 'preview only', '--agent', 'deep-research', '--dry-run'], { env: rdEnv }); + assert( + /Agent deep-research-preview-04-2026 — estimated \$1\.00–\$3\.00 for this task\. Nothing submitted \(--dry-run\)\./.test(rd), + `dry-run prints the projected spend: ${rd}` + ); + assert.strictEqual(taskCount(rdEnv.GEMCATCH_HOME), 0, 'research --dry-run submits nothing'); + ok('research --agent --dry-run prints the band and submits nothing'); + + // ---- batch --agent --dry-run: N × band total, nothing submitted ---- + const bdEnv = { GEMCATCH_HOME: path.join(HOME, 'batch-dry-agent') }; + const bdFile = path.join(HOME, 'agent-batch.txt'); + fs.writeFileSync(bdFile, 'q one\nq two\nq three\n'); + const bd = await out(['batch', bdFile, '--agent', 'deep-research-max', '--dry-run'], { env: bdEnv }); + assert( + /3 prompts × deep-research-max-preview-04-2026 — estimated \$9\.00–\$21\.00 total\. Nothing submitted \(--dry-run\)\./.test(bd), + `batch dry-run multiplies the band by N: ${bd}` + ); + assert.strictEqual(taskCount(bdEnv.GEMCATCH_HOME), 0, 'batch --dry-run submits nothing'); + ok('batch --agent --dry-run prints the N × band total and submits nothing'); + + // ---- batch --agent --yes: every row carries the resolved agent ---- + const baEnv = { GEMCATCH_HOME: path.join(HOME, 'batch-agent') }; + const ba = await cli(['batch', bdFile, '--agent', 'deep-research', '--yes', '--json', '-t', 'agents'], { env: baEnv }); + assert(/3 prompts × deep-research-preview-04-2026 — estimated \$3\.00–\$9\.00 total/.test(ba.stderr), + `the batch guard quotes N × band on stderr: ${ba.stderr}`); + const baJson = JSON.parse(ba.stdout); + assert.strictEqual(baJson.submitted.length, 3, 'all three submitted'); + const baRows = qall(baEnv.GEMCATCH_HOME, 'SELECT agent, model FROM tasks'); + for (const r of baRows) { + assert.strictEqual(r.agent, 'deep-research-preview-04-2026', 'each batch row stores the resolved agent'); + assert.strictEqual(r.model, null, 'no model on agent rows'); + } + ok('batch --agent --yes submits the file with the resolved agent on every row'); + + // ---- an agent run that hits its budget: status incomplete is terminal, the + // daemon retires it and converges instead of spinning. ---- + const bpEnv = { GEMCATCH_HOME: path.join(HOME, 'budget-pause') }; + const bpRun = await cli(['research', 'BUDGETPAUSE giant question', '--agent', 'deep-research', '--yes'], { env: bpEnv }); + const bpId = idOf(bpRun.stdout); + const bpDaemon = await cliTimeout(['daemon', '-i', '0.05', '--exit-when-idle', '--json'], { env: bpEnv }, 15000); + assert.strictEqual(bpDaemon.timedOut, false, 'the daemon must converge on an incomplete agent run, not spin'); + const bpRow = qget(bpEnv.GEMCATCH_HOME, 'SELECT status FROM tasks WHERE id = ?', bpId); + assert.strictEqual(bpRow.status, 'incomplete', `a budget pause retires the task as incomplete: ${JSON.stringify(bpRow)}`); + ok('an agent run paused by its token budget (status incomplete) retires cleanly; the daemon converges'); + + // ---- an agent interaction 404ing after the free tier's 1-day retention + // takes the existing retire-to-incomplete path unchanged. ---- + const axEnv = { GEMCATCH_HOME: path.join(HOME, 'agent-expire') }; + const axRun = await cli(['research', 'SLOW agent expiry', '--agent', 'deep-research', '--yes'], { env: axEnv }); + const axId = idOf(axRun.stdout); + const axIid = qget(axEnv.GEMCATCH_HOME, 'SELECT interaction_id FROM tasks WHERE id = ?', axId).interaction_id; + interactions.delete(axIid); + const axDaemon = await cliTimeout(['daemon', '-i', '0.05', '--exit-when-idle', '--json'], { env: axEnv }, 15000); + assert.strictEqual(axDaemon.timedOut, false, 'the daemon converges once the 404 retires the agent task'); + const axRow = qget(axEnv.GEMCATCH_HOME, 'SELECT status FROM tasks WHERE id = ?', axId); + assert.strictEqual(axRow.status, 'incomplete', 'an expired agent interaction retires to incomplete'); + ok('an expired (404) agent interaction retires to incomplete exactly like a model one'); + // ---- #3: the default SDK transport, exercised against a stubbed @google/genai. // Every test above forces GEMCATCH_FORCE_REST=1, so sdkInteractions() is // otherwise never covered. Inject a stub client and drive it directly. ---- @@ -728,8 +1012,10 @@ async function submit(prompt, args, extra) { create: async (body) => { assert.strictEqual(body.background, true, 'SDK submit must pass background:true'); assert.strictEqual(typeof body.input, 'string', 'SDK input must be a plain string'); + assert(!(body.agent && body.model), 'SDK submit must never send agent AND model'); + assert(body.agent || body.model, 'SDK submit must send one of agent or model'); const id = `sdk_${++sdkSeq}`; - sdkState.set(id, { polls: /SLOW/.test(body.input) ? 1 : 0, boom: /BOOM/.test(body.input) }); + sdkState.set(id, { polls: /SLOW/.test(body.input) ? 1 : 0, boom: /BOOM/.test(body.input), agent: body.agent }); return { id, status: 'in_progress' }; }, get: async (id) => { @@ -747,6 +1033,21 @@ async function submit(prompt, args, extra) { s.polls -= 1; return { id, status: 'in_progress' }; } + // An agent run through the SDK: multi-step, report last, citations + // attached -- shape() must read it the same way it reads REST. + if (s.agent) { + return { + id, + agent: s.agent, + status: 'completed', + usage: { total_tokens: 900 }, + steps: [ + { type: 'user_input', content: [{ type: 'text', text: 'the question' }] }, + { type: 'model_output', content: [{ type: 'text', text: 'interim notes' }] }, + { type: 'model_output', content: [{ type: 'text', text: AGENT_ANSWER }], citations: AGENT_CITATIONS }, + ], + }; + } // The SDK synthesises output_text; return one so shape() prefers it. return { id, status: 'completed', output_text: ANSWER, usage: { total_tokens: 7 } }; }, @@ -779,6 +1080,14 @@ async function submit(prompt, args, extra) { assert(/Unknown model id via SDK/.test(boom.message), `friendly() must surface the SDK's real payload: ${boom.message}`); assert.strictEqual(boom.httpStatus, 400, 'friendly() maps the SDK error to its HTTP status'); + // The agent path through the SDK: `agent` on create, no `model`, and the + // final-step report with citations coming back through shape(). + const sa = await sdk.submit('deep dive', { agent: 'deep-research-preview-04-2026' }); + const sp = await sdk.poll(sa.interactionId); + assert.strictEqual(sp.status, 'completed'); + assert.strictEqual(sp.text, AGENT_ANSWER, 'the SDK agent run yields only the final-step report'); + assert.deepStrictEqual(sp.citations, AGENT_CITATIONS, 'citations survive the SDK path'); + // Restore: nothing after this should see the stub or the fake key. delete require.cache[geminiPath]; delete require.cache[genaiPath]; From 233ab088ccbaac7ed4d0247197a6f96dee0b12c9 Mon Sep 17 00:00:00 2001 From: Booyaka101 Date: Sat, 8 Aug 2026 09:02:44 +0800 Subject: [PATCH 3/4] docs: PROGRESS.md for the 0.4.0 release state Co-Authored-By: Claude Fable 5 --- PROGRESS.md | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) create mode 100644 PROGRESS.md diff --git a/PROGRESS.md b/PROGRESS.md new file mode 100644 index 0000000..17c91b0 --- /dev/null +++ b/PROGRESS.md @@ -0,0 +1,24 @@ +# PROGRESS — gemcatch 0.4.0 (Deep Research agents) + +**State: COMPLETE.** Branch `feat/deep-research-agents` (off `fix/better-sqlite3-13`, which carries the better-sqlite3 13 bump not yet on `main`), commit `c5869cc`. Version bumped 0.3.0 → 0.4.0, CHANGELOG entry written, README has the new "Research agents" section. + +## Verified working (all offline, `npm test`, 64 tests green — 49 pre-existing + 15 new) + +- `research`/`batch -a/--agent`: `agent` sent INSTEAD of `model` on create (both SDK-stub and REST paths asserted); aliases resolve through the one table in `gemini.js` (`deep-research` → `deep-research-preview-04-2026`, `deep-research-max` → `…-max-preview-04-2026`); unknown ids pass through and the API's 4xx surfaces unretried; `--model` + `--agent` is a clean error with zero rows written. +- Spend guard: band printed before every agent submission ($1.00–$3.00 / $3.00–$7.00, hedged "preview rates, subject to change"); batch prints N × band; interactive y/N (test hook `GEMCATCH_ASSUME_TTY=1`); non-TTY without `--yes` refused; decline → zero rows, exit 1; `--dry-run` on research and batch prints projected spend, submits nothing. +- Extraction: final answer-bearing step (docs: `steps[-1].content[0].text`) with collect-all fallback; interim agent steps never leak; citations persisted (new column), printed as `Sources:`, in `--json`. +- Migration: v1 and v0.3.0 `tasks.db` upgrade in place, all rows kept, `agent` NULL for old rows. +- `incomplete` (max_total_tokens budget pause) and agent-404-after-retention both retire cleanly; daemon converges. +- Default model now `gemini-3.5-flash-lite`. +- Packaging: `npm pack` → tarball installed in a clean scratch dir, `.bin/gemcatch --version` → 0.4.0; full worked-example session driven for real against a standalone mock (confirm-y submit → batch dry-run → daemon → get with sources → list AGENT column → stats agent tally). + +## Phase 0 (all re-verified live 2026-08-08) + +deep-research doc (agent ids, `agent=` vs `model=`, background mandatory, `steps[-1]`, citations, price bands + hedge); interactions-overview (55 days paid / 1 day free, three agents listed); blog 2026-07-28 (free-tier managed agents; `max_total_tokens` → `status: "incomplete"`); changelog (3.5-flash-lite GA 2026-07-21). + +## Next steps (owner, from the phone) + +1. Merge `fix/better-sqlite3-13` → `main` (PR #? — the dependabot-adjacent branch), then PR `feat/deep-research-agents` → `main`. +2. `npm publish` (prepublishOnly runs the suite) + `git tag v0.4.0` + GitHub release — per RELEASING.md. +3. Optional live smoke on a real key: `gemcatch research "test" -w` (free model, $0) and `gemcatch research "…" --agent deep-research --dry-run` ($0). A real agent run costs $1–$3 — owner's call. +4. Distribution: the 0.4.0 story writes itself — "the docs require background execution for Deep Research agents; gemcatch already was that client" (dev.to per the usual channel playbook). From 91558f2a16bbbd9d3a0bec65e86176dd6159745f Mon Sep 17 00:00:00 2001 From: Booyaka101 Date: Sat, 8 Aug 2026 09:12:58 +0800 Subject: [PATCH 4/4] ci: install with --ignore-scripts, so a prebuilt dep is not compiled better-sqlite3 13.x dropped its `install` script and ships N-API prebuilds in the tarball, so nothing in the tree needs building. Without --ignore-scripts npm sees its binding.gyp, falls back to `node-gyp rebuild`, and node 22 / windows-latest fails: the runner image now carries Visual Studio 18, which node-gyp 11.5.0 (bundled with Node 22) reports as an unknown version. Node 24 ships a newer node-gyp and passed, which is why only one matrix cell was red. Verified on Windows + Node 22 (the failing combination): npm ci --ignore-scripts leaves no build/Release, loads prebuilds/win32-x64.node, and the full suite passes. The package job still does a real, un-ignored global install from the packed tarball, so the user install path stays covered. Co-Authored-By: Claude Fable 5 --- .github/workflows/ci.yml | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 47effd1..d1d89cc 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -23,7 +23,14 @@ jobs: node-version: ${{ matrix.node }} cache: npm - - run: npm ci + # --ignore-scripts: better-sqlite3 13.x dropped its `install` script and + # ships N-API prebuilds in the tarball (prebuilds/-.node), + # so nothing here needs building. Without this, npm sees the package's + # binding.gyp, falls back to `node-gyp rebuild`, and the build fails on + # windows-latest images carrying Visual Studio 18 -- node-gyp 11.5.0 (the + # one bundled with Node 22) reports it as "unknown version undefined" and + # gives up. Nothing in the dependency tree needs an install script to run. + - run: npm ci --ignore-scripts # The suite runs the real CLI against a mock Interactions API on # localhost. No GEMINI_API_KEY is needed, so this is safe on forks. @@ -40,7 +47,7 @@ jobs: node-version: '22' cache: npm - - run: npm ci + - run: npm ci --ignore-scripts # Catches a broken bin entry or a missing file in `files` before a # release does.