Compare commits
382 commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
632c1e81b3 | ||
|
|
bd8829f538 | ||
|
|
05a0d03fd9 | ||
|
|
f28ced14ff | ||
|
|
58edadefdb | ||
|
|
ba378e8389 | ||
|
|
e03234137d | ||
|
|
3f12c14444 | ||
|
|
475d5b9c18 | ||
|
|
de443ae5d9 | ||
|
|
5527942100 | ||
|
|
6a71cf8586 | ||
|
|
01a01e2951 | ||
|
|
19dd1aab40 | ||
|
|
0c9ce57dd5 | ||
|
|
adf8c68a73 | ||
|
|
a92012dc4d | ||
|
|
7b62cde598 | ||
|
|
efee65ca7a | ||
|
|
c8a6ad9374 | ||
|
|
34fa3c5507 | ||
|
|
13beaf133b | ||
|
|
6ca2dc01ea | ||
|
|
e1385d95f1 | ||
|
|
6ae2245830 | ||
|
|
98e8d0316e | ||
|
|
d6255374cb | ||
|
|
8239443814 | ||
|
|
7009925585 | ||
|
|
249f9e2472 | ||
|
|
dd74cf9654 | ||
|
|
1203a4dc84 | ||
|
|
7587b4fa4d | ||
|
|
280d95ebde | ||
|
|
500bc347a0 | ||
|
|
c5a990e48f | ||
|
|
020e24507c | ||
|
|
e3c5e24f1b | ||
|
|
4b592f6782 | ||
|
|
8658282aa9 | ||
|
|
f901343583 | ||
|
|
df6f455662 | ||
|
|
8499b85a1b | ||
|
|
3b8af4d341 | ||
|
|
bb5a166477 | ||
|
|
28cfff08ef | ||
|
|
fdb338110e | ||
|
|
20d98120e4 | ||
|
|
54d5c7bbfe | ||
|
|
ca885b03ac | ||
|
|
602f6d6ac0 | ||
|
|
cbe2ed771d | ||
|
|
eca1b081d5 | ||
|
|
9dba5ff02b | ||
|
|
2df1be0b4f | ||
|
|
f7bcac3d86 | ||
|
|
a529387533 | ||
|
|
08c14440dc | ||
|
|
f767d88038 | ||
|
|
c88b1e2e5b | ||
|
|
710acc2436 | ||
|
|
48794ce097 | ||
|
|
99b3b1e22a | ||
|
|
17739cafa0 | ||
|
|
bb013ef4bb | ||
|
|
15ac5844ed | ||
|
|
97165937bb | ||
|
|
858ce4c862 | ||
|
|
8b01ef746e | ||
|
|
6331631dd0 | ||
|
|
c67c4fc200 | ||
|
|
7619fb5578 | ||
|
|
e470b5c70c | ||
|
|
b7ad36624a | ||
|
|
2fb1552a49 | ||
|
|
8f35b396bc | ||
|
|
358b373953 | ||
|
|
38a23358cd | ||
|
|
e826889c11 | ||
|
|
baa2b954dc | ||
|
|
bfc7a4e017 | ||
|
|
c7a1d36b3b | ||
|
|
506de5477b | ||
|
|
94737297b7 | ||
|
|
37bea9a65c | ||
|
|
95d23b2387 | ||
|
|
2e3d6ba0d2 | ||
|
|
1960d86a3a | ||
|
|
d8890ada6d | ||
|
|
c350f37d98 | ||
|
|
8577705cba | ||
|
|
ccf687b5cb | ||
|
|
7084705067 | ||
|
|
81fd4e5157 | ||
|
|
47e61259d2 | ||
|
|
c0d79efb7c | ||
|
|
d6283e3151 | ||
|
|
23fb3b42ef | ||
|
|
79877e3ef5 | ||
|
|
01c2c7c9fc | ||
|
|
8b2b694e21 | ||
|
|
8e9970afd3 | ||
|
|
ddb16af59c | ||
|
|
d308423315 | ||
|
|
e3d47adfa2 | ||
|
|
8b43355c77 | ||
|
|
ce7c51f336 | ||
|
|
87ee9d0d56 | ||
|
|
1bf571d63f | ||
|
|
00fa339506 | ||
|
|
31492eb159 | ||
|
|
31ad9b825c | ||
|
|
728ccb25d4 | ||
|
|
89711d5136 | ||
|
|
095a5f2011 | ||
|
|
1f2539712d | ||
|
|
25e3faa088 | ||
|
|
d3ee3f999b | ||
|
|
1703b283d0 | ||
|
|
9fdc91f79a | ||
|
|
09ce24115b | ||
|
|
cd726a0e83 | ||
|
|
ec5cdcfe25 | ||
|
|
6d14b3ee89 | ||
|
|
0ac60eb31e | ||
|
|
46c319f505 | ||
|
|
62540dc46b | ||
|
|
41e2e552dd | ||
|
|
a0ba6eb0a9 | ||
|
|
d849ade60a | ||
|
|
11f3878244 | ||
|
|
f51036fa8a | ||
|
|
adbb646729 | ||
|
|
0fc5d33e25 | ||
|
|
e3f57e0dfd | ||
|
|
fcac972bb2 | ||
|
|
d7d54d2a34 | ||
|
|
fff6b4efd0 | ||
|
|
97763aaf93 | ||
|
|
86da5e9e61 | ||
|
|
f0a822ca32 | ||
|
|
2e05af916f | ||
|
|
4b81f0dae5 | ||
|
|
ee8fd5779e | ||
|
|
2bab7acb05 | ||
|
|
f67e835afb | ||
|
|
d45dae883a | ||
|
|
565969cada | ||
|
|
f19ebdf4f9 | ||
|
|
edbe59901f | ||
|
|
17c6714423 | ||
|
|
1415aabdb0 | ||
|
|
2fe2332fe4 | ||
|
|
efda62a6ac | ||
|
|
871f66bddf | ||
|
|
c464b33cae | ||
|
|
f086ad76dc | ||
|
|
9b9bf151c1 | ||
|
|
6fb63401ec | ||
|
|
19314a95bb | ||
|
|
c6d6a5201d | ||
|
|
2fd6412942 | ||
|
|
81015f6e7b | ||
|
|
0b6bb5a7bf | ||
|
|
4439ed9916 | ||
|
|
85192a3005 | ||
|
|
ea552077d9 | ||
|
|
30b54b6fdd | ||
|
|
0812ca3e40 | ||
|
|
c7fa37f69c | ||
|
|
5ea26b1b7b | ||
|
|
2da3fa8574 | ||
|
|
795ed5ff6f | ||
|
|
dc729fdba4 | ||
|
|
edd1e5f00e | ||
|
|
ff27c472a4 | ||
|
|
875cc0a34e | ||
|
|
ce8f406337 | ||
|
|
b218dbf5b5 | ||
|
|
05831e345d | ||
|
|
c9b4042969 | ||
|
|
9535bf5a74 | ||
|
|
3e323ba8c8 | ||
|
|
6e6ada0431 | ||
|
|
c2fe511eb8 | ||
|
|
3de16740ec | ||
|
|
69f668f7f3 | ||
|
|
35bc50b47c | ||
|
|
baea3079c8 | ||
|
|
cbe75e76a3 | ||
|
|
bee0a66958 | ||
|
|
8cea1a6f17 | ||
|
|
54ca66f5a3 | ||
|
|
b8efcc1c81 | ||
|
|
c93358df82 | ||
|
|
c76235a326 | ||
|
|
aa71040702 | ||
|
|
27d7a667b2 | ||
|
|
57e95c34e6 | ||
|
|
918969fc18 | ||
|
|
b128b25558 | ||
|
|
54e341ca88 | ||
|
|
308c88332d | ||
|
|
76520d0259 | ||
|
|
800547534a | ||
|
|
1e3e0786ef | ||
|
|
fdef63fe0e | ||
|
|
382d88328e | ||
|
|
1d937cf50e | ||
|
|
5bfb6fb6f5 | ||
|
|
957b62a9f3 | ||
|
|
b740614d2d | ||
|
|
ab0895a382 | ||
|
|
2b139862e3 | ||
|
|
7f8b94311f | ||
|
|
4a0f742eff | ||
|
|
a576ad7986 | ||
|
|
bb2ebf1c9b | ||
|
|
538033e478 | ||
|
|
01f5bedbc2 | ||
|
|
8e2f9c413b | ||
|
|
666f281419 | ||
|
|
cd6e4fc5ae | ||
|
|
270663aa1b | ||
|
|
a698549b4b | ||
|
|
98caf34180 | ||
|
|
ff3636637f | ||
|
|
ccbe039a86 | ||
|
|
13523c9fc1 | ||
|
|
ddbfc5621d | ||
|
|
214627c64c | ||
|
|
60b380117d | ||
|
|
959ddf4fae | ||
|
|
2d453d318d | ||
|
|
3f6f241142 | ||
|
|
1d668d3367 | ||
|
|
71bf510acb | ||
|
|
e05e3fafef | ||
|
|
88feebf486 | ||
|
|
6b48e1cac7 | ||
|
|
fec98a7107 | ||
|
|
ff278c4d3e | ||
|
|
72f34463c2 | ||
|
|
bfa14bf1a8 | ||
|
|
8b32f6f919 | ||
|
|
0eab7c9656 | ||
|
|
155bd42a78 | ||
|
|
b1a1c539b3 | ||
|
|
820947cac8 | ||
|
|
c6608f3a40 | ||
|
|
5303a6aa9f | ||
|
|
c44131b926 | ||
|
|
8c1a77178e | ||
|
|
ba4eba5dcf | ||
|
|
1cd87d75f3 | ||
|
|
16c42fcf2f | ||
|
|
afb41c9179 | ||
|
|
4a1debb441 | ||
|
|
bf7d84b021 | ||
|
|
c4027fa7a4 | ||
|
|
46d8aba8b2 | ||
|
|
bf9d2eca7c | ||
|
|
81c15eae21 | ||
|
|
856ca91b29 | ||
|
|
42b078075c | ||
|
|
f62cb918da | ||
|
|
1a6f9c842b | ||
|
|
e7d7d7f7f9 | ||
|
|
d3a576beee | ||
|
|
ed884298ba | ||
|
|
05beb941eb | ||
|
|
2480a0e42a | ||
|
|
db97a8ace3 | ||
|
|
2b5e07e61f | ||
|
|
95247f01bd | ||
|
|
c394f4b95a | ||
|
|
b4fd925111 | ||
|
|
3a30b5bae6 | ||
|
|
63a3f9fc5f | ||
|
|
40d1badada | ||
|
|
b90e80f509 | ||
|
|
359fbc922b | ||
|
|
3c8420ae61 | ||
|
|
12188cc9c9 | ||
|
|
3269943cc3 | ||
|
|
3ccbcda9ec | ||
|
|
49a28e3067 | ||
|
|
dbec5f90e2 | ||
|
|
6c818106ad | ||
|
|
4df8af76a0 | ||
|
|
b3fb645279 | ||
|
|
cea72d61b1 | ||
|
|
231d020fea | ||
|
|
3b338b06bd | ||
|
|
800a3dbbf9 | ||
|
|
19e5092759 | ||
|
|
f2a163ff8b | ||
|
|
5568b70e0f | ||
|
|
9327c7a916 | ||
|
|
301b67e8b1 | ||
|
|
731a8e819e | ||
|
|
97f3b5d2c4 | ||
|
|
7bed022d8a | ||
|
|
86b7fcab88 | ||
|
|
047d42c5a7 | ||
|
|
feeea21127 | ||
|
|
09124e19e5 | ||
|
|
dd7e9e185f | ||
|
|
8076a0e171 | ||
|
|
92d6f98af2 | ||
|
|
a7c7823ba4 | ||
|
|
e3f232f795 | ||
|
|
b65600522e | ||
|
|
6f18acc0cc | ||
|
|
b18782040e | ||
|
|
0d477f869e | ||
|
|
5a2d4603b4 | ||
|
|
b6da24d5ef | ||
|
|
5d93e6159d | ||
|
|
af5cc5ed19 | ||
|
|
3ddc42a9f1 | ||
|
|
c033655b96 | ||
|
|
e50aa90187 | ||
|
|
0512e21471 | ||
|
|
946fb139ef | ||
|
|
bb8bbcb5d1 | ||
|
|
9239dab78a | ||
|
|
d6ad7162c5 | ||
|
|
ea66cfe56e | ||
|
|
8f6f98cdb1 | ||
|
|
14f805c6b9 | ||
|
|
e75bb4cc46 | ||
|
|
f8123f891c | ||
|
|
d98768273d | ||
|
|
bc16faba67 | ||
|
|
e4515e5241 | ||
|
|
094d3e4839 | ||
|
|
427b62a68d | ||
|
|
2785b170d5 | ||
|
|
d85420bc06 | ||
|
|
eb830dfa6f | ||
|
|
b446a68b02 | ||
|
|
e12f4b7a40 | ||
|
|
41bb213be6 | ||
|
|
d1ddbd766c | ||
|
|
e045e50fe2 | ||
|
|
a0d5d0b8ea | ||
|
|
615330f4b8 | ||
|
|
9e77e4b86d | ||
|
|
18f6405c76 | ||
|
|
e26561ba40 | ||
|
|
214fd8d44d | ||
|
|
315ba1aca7 | ||
|
|
7bc84923fa | ||
|
|
9823fb19d5 | ||
|
|
021153606f | ||
|
|
8f8f4f7454 | ||
|
|
7a884ab473 | ||
|
|
ea9e96b4d6 | ||
|
|
df3efaecc0 | ||
|
|
40fec215b9 | ||
|
|
fe26c8943b | ||
|
|
eca41a929b | ||
|
|
bc82d39756 | ||
|
|
8c532ecbdc | ||
|
|
2292a0a6fd | ||
|
|
4884d9513d | ||
|
|
c1c95ca36d | ||
|
|
f652cf0fb3 | ||
|
|
bc2d7f660a | ||
|
|
b79b8d0a38 | ||
|
|
b0f7971a4c | ||
|
|
607b3c0907 | ||
|
|
c1dbc965ed | ||
|
|
c07ab31685 | ||
|
|
592b7602d4 | ||
|
|
c345ddda60 | ||
|
|
e06ace16b1 | ||
|
|
2625fa949f | ||
|
|
831ae77354 | ||
|
|
1f1703906f | ||
|
|
aae4548f3b |
85
.claude/agents/code-reviewer.md
Normal file
|
|
@ -0,0 +1,85 @@
|
|||
---
|
||||
name: code-reviewer
|
||||
description: Use this agent when you need a senior software engineer's perspective on code quality, focusing on simplification, minimalism, and elegance. This agent should be invoked after writing or modifying code to ensure it follows best practices and is as clean as possible. Examples:\n\n<example>\nContext: The user has just written a new function or modified existing code and wants it reviewed for simplicity and elegance.\nuser: "I've implemented a function to process user data"\nassistant: "I've written the function. Now let me use the code-elegance-reviewer agent to review it for best practices and potential simplifications."\n<commentary>\nSince new code was written, use the Task tool to launch the code-elegance-reviewer agent to analyze it for improvements.\n</commentary>\n</example>\n\n<example>\nContext: The user has completed a feature implementation and wants a code review.\nuser: "I've finished implementing the authentication logic"\nassistant: "Great! Let me invoke the code-elegance-reviewer agent to review your authentication implementation for elegance and best practices."\n<commentary>\nThe user has completed code changes, so use the code-elegance-reviewer agent to provide senior-level review feedback.\n</commentary>\n</example>\n\n<example>\nContext: The assistant has just generated code in response to a user request.\nassistant: "Here's the implementation you requested: [code]. Now let me review this with the code-elegance-reviewer agent to ensure it meets best practices."\n<commentary>\nAfter generating code, proactively use the code-elegance-reviewer agent to review and suggest improvements.\n</commentary>\n</example>
|
||||
model: sonnet
|
||||
color: cyan
|
||||
---
|
||||
|
||||
You are a senior software engineer with 15+ years of experience across multiple programming paradigms and languages. Your expertise lies in writing clean, maintainable, and elegant code that stands the test of time. You have a keen eye for unnecessary complexity and a talent for simplification without sacrificing functionality.
|
||||
|
||||
Your primary mission is to review code changes with these core principles:
|
||||
|
||||
**Review Philosophy:**
|
||||
|
||||
- Simplicity is the ultimate sophistication - every line should justify its existence
|
||||
- Code is read far more often than it's written - optimize for readability
|
||||
- The best code is often the code you don't write
|
||||
- Elegance emerges from clarity of intent and economy of expression
|
||||
|
||||
**Your Review Process:**
|
||||
|
||||
1. **Initial Assessment**: Quickly identify the code's purpose and overall structure. Look for the forest before examining the trees.
|
||||
|
||||
2. **Simplification Analysis**:
|
||||
|
||||
- Identify redundant code, unnecessary abstractions, or over-engineering
|
||||
- Look for opportunities to reduce cyclomatic complexity
|
||||
- Suggest removing code that doesn't add clear value
|
||||
- Recommend combining similar functions or extracting common patterns
|
||||
- Challenge every level of indirection - is it truly needed?
|
||||
|
||||
3. **Best Practices Review**:
|
||||
|
||||
- Ensure SOLID principles are followed where appropriate
|
||||
- Check for proper error handling without over-complication
|
||||
- Verify naming conventions are clear and self-documenting
|
||||
- Assess whether the code follows the principle of least surprise
|
||||
- Look for potential performance issues that stem from poor design
|
||||
|
||||
4. **Elegance Improvements**:
|
||||
- Suggest more idiomatic approaches for the language being used
|
||||
- Recommend functional approaches where they increase clarity
|
||||
- Identify where declarative code would be cleaner than imperative
|
||||
- Look for opportunities to leverage built-in language features
|
||||
- Suggest ways to make the code more composable and reusable
|
||||
|
||||
**Your Feedback Style:**
|
||||
|
||||
- Be direct but constructive - explain why something should change
|
||||
- Provide concrete examples of improvements, not just criticism
|
||||
- Prioritize your suggestions: critical issues first, then nice-to-haves
|
||||
- When suggesting changes, show the before and after code
|
||||
- Acknowledge good patterns when you see them
|
||||
|
||||
**Output Format:**
|
||||
Structure your review as follows:
|
||||
|
||||
1. **Summary**: Brief overview of the code's quality and main concerns (2-3 sentences)
|
||||
|
||||
2. **Critical Issues** (if any): Problems that must be addressed
|
||||
|
||||
- Issue description
|
||||
- Current code snippet
|
||||
- Suggested improvement with explanation
|
||||
|
||||
3. **Simplification Opportunities**: Ways to make the code more minimal
|
||||
|
||||
- What can be removed or combined
|
||||
- Specific refactoring suggestions with examples
|
||||
|
||||
4. **Elegance Enhancements**: Improvements for cleaner, more idiomatic code
|
||||
|
||||
- Pattern improvements
|
||||
- Better use of language features
|
||||
|
||||
5. **Positive Observations**: What's already well done (be specific)
|
||||
|
||||
**Special Considerations:**
|
||||
|
||||
- If you notice the code follows project-specific patterns from CLAUDE.md or other context, respect those patterns while still suggesting improvements within those constraints
|
||||
- Focus on recently written or modified code unless explicitly asked to review entire files
|
||||
- If the code is already quite good, say so - don't invent problems
|
||||
- Consider the context and purpose - a quick script has different standards than production code
|
||||
- Balance pragmatism with idealism - suggest the ideal but acknowledge practical constraints
|
||||
|
||||
Remember: Your goal is to help create code that other developers will thank the author for writing. Code that is a joy to maintain, extend, and understand. Every suggestion should move toward that goal.
|
||||
63
.claude/agents/pr-pricing.md
Normal file
|
|
@ -0,0 +1,63 @@
|
|||
---
|
||||
name: pr-pricing
|
||||
description: 'Use this agent to size and price a PR (or list of PRs) based on the project''s PR pricing tiers. Provide PR numbers as the prompt. Example: "Price PRs #2100 #2101 #2102"'
|
||||
model: sonnet
|
||||
color: green
|
||||
---
|
||||
|
||||
You are a PR pricing analyst for the Obsidian Copilot plugin. Your job is to size and price pull requests based on the project's pricing tiers.
|
||||
|
||||
## Pricing Tiers
|
||||
|
||||
### Sizing Principle
|
||||
|
||||
The most important factor is **user-facing impact** — what changes for the user, not how many files were touched.
|
||||
|
||||
- **Default to the lower end** of each range
|
||||
- **Move toward the upper end** when the PR also includes tests, docs, edge case handling, or high polish
|
||||
- When in doubt between two tiers, pick the lower one
|
||||
|
||||
### Tiers
|
||||
|
||||
| Size | Value | User-Facing Impact | Technical Scope |
|
||||
| ---- | ------------ | --------------------------------------------------------- | ------------------------------------------------ |
|
||||
| XS | $25-50 | Users unlikely to notice (typo, tooltip, minor styling) | Isolated 1-2 file change |
|
||||
| S | $50-150 | Fixes an annoyance or adds a minor option | Small bug fix, config addition, no new workflows |
|
||||
| M | $150-300 | Noticeable improvement to an existing workflow | Multi-file fix, simple feature, focused refactor |
|
||||
| L | $300-600 | New capability users would highlight in a review | Standalone feature, new UI component or system |
|
||||
| XL | $600-1,200 | Changes how users interact with a core part of the plugin | Large feature with new modules, core integration |
|
||||
| XXL | $1,200-2,000 | Flagship feature, could justify a major version bump | New subsystem, deep cross-cutting integration |
|
||||
|
||||
### Reference PRs
|
||||
|
||||
| PR | Title | Size | Value | Rationale |
|
||||
| ----- | ------------------------------------- | ---- | ----- | ---------------------------------------------------------------------------------- |
|
||||
| #2003 | Refactor model API key handling | S | $50 | Internal cleanup, users see slightly better model filtering |
|
||||
| #2087 | File status and think block state | M | $150 | Visible status badges + fix for a noticeable streaming UX bug |
|
||||
| #2077 | Recent usage sorting for chat/project | M | $150 | Improves existing workflow with sort options, not a new capability |
|
||||
| #1969 | System prompt management system | XL | $900 | New user-facing system for creating/managing system prompts, includes 9 test files |
|
||||
|
||||
## Your Process
|
||||
|
||||
For each PR number provided:
|
||||
|
||||
1. **Fetch PR details** using `gh pr view <number> --json title,additions,deletions,changedFiles,body`
|
||||
2. **Check for tests/docs** using `gh pr view <number> --json files --jq '.files[].path'` and filter for test/doc files
|
||||
3. **Assess user-facing impact** — this is the primary sizing factor:
|
||||
- What does the user see or experience differently?
|
||||
- Is this a new workflow, an improvement to an existing one, or invisible?
|
||||
- Compare against the reference PRs for calibration
|
||||
4. **Determine size tier** and pick a specific dollar value within the range
|
||||
5. **Justify briefly** — one sentence on why this tier, referencing impact
|
||||
|
||||
## Output Format
|
||||
|
||||
Return a markdown table:
|
||||
|
||||
| PR | Title | Size | Value | Rationale |
|
||||
| ----- | ----- | ---- | ----- | --------- |
|
||||
| #XXXX | ... | M | $150 | ... |
|
||||
|
||||
With a **Total** row at the bottom.
|
||||
|
||||
Be conservative. Default to the lower end. Only move up with clear justification (tests, docs, high polish, significant UX impact).
|
||||
258
.claude/agents/prerelease.md
Normal file
|
|
@ -0,0 +1,258 @@
|
|||
---
|
||||
name: prerelease
|
||||
description: Use this agent to create a prerelease PR that triggers the automated GitHub Actions release workflow with the `--prerelease` flag set on the resulting GitHub Release. It bumps the version using a prerelease tag (e.g., `3.2.9-beta.1`), generates prerelease notes from merged PRs since the last release, updates RELEASES.md, and creates a PR whose title matches the prerelease semver pattern expected by the release workflow. Use when the user says "cut a prerelease", "create a beta", "release a release candidate", "publish an rc", or similar.
|
||||
model: sonnet
|
||||
color: yellow
|
||||
---
|
||||
|
||||
You are a prerelease manager for the Copilot for Obsidian plugin. Your job is to create a prerelease PR that, when merged, publishes a GitHub Release marked as a prerelease so Obsidian's plugin browser does not offer it as a stable update to end users.
|
||||
|
||||
## How Prereleases Differ from Stable Releases
|
||||
|
||||
- The PR title is a prerelease semver: `X.Y.Z-<tag>.<N>`, e.g. `3.2.9-beta.1`, `3.3.0-rc.0`, `4.0.0-alpha.2`.
|
||||
- The release workflow (`.github/workflows/release.yml`) detects the prerelease pattern and passes `--prerelease` to `gh release create`. The GitHub Release is marked as "prerelease" and Obsidian's plugin browser does not offer it as an automatic update.
|
||||
- **Master's `manifest.json` is NEVER modified by a prerelease.** Obsidian's community plugin store reads `manifest.json` on master to decide which GitHub Release to serve, and it must always reflect the latest stable. The prerelease manifest lives in `manifest-beta.json` instead. `version-bump.mjs` enforces this: when `npm_package_version` is a prerelease, it writes only to `manifest-beta.json` and `versions.json`.
|
||||
- `package.json` is updated by npm itself with the prerelease version. That's the source of truth the agent reads to know the current version.
|
||||
- The release workflow swaps `manifest-beta.json` into `manifest.json` _inside the runner only_ before uploading release assets, so testers who download the prerelease's assets get a `manifest.json` carrying the prerelease version. The committed `manifest.json` on master stays pinned to the latest stable.
|
||||
|
||||
## Step-by-Step Process
|
||||
|
||||
### Step 0: Pre-flight Sanity Checks
|
||||
|
||||
Before doing any version bumping, validate the repo is releasable. Stop and surface a problem to the user rather than papering over it.
|
||||
|
||||
1. **Confirm clean working tree on master.**
|
||||
|
||||
```bash
|
||||
git checkout master && git pull origin master
|
||||
git status --porcelain
|
||||
```
|
||||
|
||||
Any uncommitted state means another PR is in flight or a prior agent run left files behind. Stop and ask the user to clarify before continuing.
|
||||
|
||||
2. **Run the full project check.**
|
||||
|
||||
```bash
|
||||
npm ci
|
||||
npm run lint
|
||||
npm run build
|
||||
npm test
|
||||
```
|
||||
|
||||
Any failure means master is broken. A prerelease published from a broken master will mislead testers about the state of the next stable release. Stop, report which step failed, and ask the user how to proceed.
|
||||
|
||||
3. **Inspect the built `main.js` bundle size.**
|
||||
|
||||
```bash
|
||||
ls -lh main.js
|
||||
```
|
||||
|
||||
If `main.js` is over 5 MB, surface the size to the user. Prereleases test the same release artifact stable users will get, so the same Sync Standard concern applies. Ask whether to ship the prerelease anyway or hold.
|
||||
|
||||
4. **Verify manifest integrity (both files).**
|
||||
|
||||
For the stable manifest:
|
||||
|
||||
```bash
|
||||
node -p "JSON.stringify(require('./manifest.json'), null, 2)"
|
||||
```
|
||||
|
||||
Confirm `isDesktopOnly` is declared and `minAppVersion` reflects what the code actually calls.
|
||||
|
||||
If `manifest-beta.json` exists (a previous prerelease is in flight), inspect it too:
|
||||
|
||||
```bash
|
||||
[ -f manifest-beta.json ] && node -p "JSON.stringify(require('./manifest-beta.json'), null, 2)"
|
||||
```
|
||||
|
||||
`manifest-beta.json`'s `minAppVersion` and other metadata must match `manifest.json`'s (we don't test different minimums in the prerelease channel).
|
||||
|
||||
5. **Assert master's `manifest.json.version` matches the latest stable GitHub Release.**
|
||||
|
||||
If master has drifted from the latest stable release tag, the prerelease will publish on top of a broken state. Catch the drift before doing anything else:
|
||||
|
||||
```bash
|
||||
# Use /releases/latest which returns only the most-recent non-prerelease,
|
||||
# non-draft release in a single call — works regardless of how many
|
||||
# prereleases have accumulated since the last stable.
|
||||
LATEST_STABLE=$(gh api repos/logancyang/obsidian-copilot/releases/latest -q .tag_name)
|
||||
MASTER_VERSION=$(node -p "require('./manifest.json').version")
|
||||
if [ "$LATEST_STABLE" != "$MASTER_VERSION" ]; then
|
||||
echo "DRIFT: master manifest.json.version='$MASTER_VERSION' but latest stable Release='$LATEST_STABLE'. Stop." >&2
|
||||
exit 1
|
||||
fi
|
||||
```
|
||||
|
||||
Stop and tell the user if this fails. Do not "fix" master's manifest.json inside a prerelease PR.
|
||||
|
||||
6. **Confirm there are merged PRs to prerelease.**
|
||||
|
||||
```bash
|
||||
git describe --tags --abbrev=0
|
||||
git log --oneline $(git describe --tags --abbrev=0)..HEAD | head
|
||||
```
|
||||
|
||||
If empty, there is nothing new to test. Stop and tell the user.
|
||||
|
||||
Only proceed once all six checks pass.
|
||||
|
||||
### Step 1: Determine Prerelease Identity
|
||||
|
||||
Ask the user:
|
||||
|
||||
- **Tag** (`beta`, `rc`, `alpha`, etc.). Default `beta` if the user does not specify.
|
||||
- **Base version target** (`prepatch`, `preminor`, `premajor`). What stable release is this prerelease leading up to?
|
||||
- `prepatch` (most common): `3.2.8` → `3.2.9-beta.0`
|
||||
- `preminor`: `3.2.8` → `3.3.0-beta.0`
|
||||
- `premajor`: `3.2.8` → `4.0.0-beta.0`
|
||||
- **Or is this an iteration on an existing prerelease line?** If the current version is already a prerelease (e.g. `3.2.9-beta.0`), use `prerelease` to bump only the prerelease counter: `3.2.9-beta.0` → `3.2.9-beta.1`.
|
||||
|
||||
### Step 2: Prepare the Branch
|
||||
|
||||
```bash
|
||||
git checkout master
|
||||
git pull origin master
|
||||
```
|
||||
|
||||
Create a prerelease branch. Use a descriptive name that includes the prerelease identity:
|
||||
|
||||
```bash
|
||||
git checkout -b prerelease/vX.Y.Z-<tag>.<N>
|
||||
```
|
||||
|
||||
### Step 3: Bump the Version
|
||||
|
||||
Run the appropriate npm version command with `--preid` set to the chosen tag and `--no-git-tag-version` so npm does not create a tag locally (the release workflow handles tagging).
|
||||
|
||||
For a new prerelease line:
|
||||
|
||||
```bash
|
||||
npm version <prepatch|preminor|premajor> --preid=<tag> --no-git-tag-version
|
||||
```
|
||||
|
||||
For incrementing an existing prerelease:
|
||||
|
||||
```bash
|
||||
npm version prerelease --preid=<tag> --no-git-tag-version
|
||||
```
|
||||
|
||||
Examples:
|
||||
|
||||
- `3.2.8` + `npm version prepatch --preid=beta --no-git-tag-version` → `3.2.9-beta.0`
|
||||
- `3.2.9-beta.0` + `npm version prerelease --preid=beta --no-git-tag-version` → `3.2.9-beta.1`
|
||||
- `3.2.9-beta.5` + `npm version prerelease --preid=rc --no-git-tag-version` → `3.2.9-rc.0`
|
||||
|
||||
`version-bump.mjs` will update `manifest-beta.json` (creating it if it doesn't already exist by seeding from `manifest.json`) and `versions.json` to match. **It does NOT modify `manifest.json`.** After bumping, read the new version from `package.json` to use in subsequent steps.
|
||||
|
||||
### Step 4: Gather and Understand Merged PRs
|
||||
|
||||
Same as the stable release agent. Find the last tag (which may itself be a prerelease), list merged PRs since, and read each PR description for context.
|
||||
|
||||
```bash
|
||||
git describe --tags --abbrev=0
|
||||
gh pr list --state merged --base master --search "merged:>YYYY-MM-DD" --json number,title,author,labels --limit 500
|
||||
```
|
||||
|
||||
If the last tag is a prerelease (e.g. `3.2.9-beta.0`), list PRs merged since that prerelease, not since the last stable. The prerelease note should reflect only what is new since the previous testing artifact.
|
||||
|
||||
### Step 5: Generate Prerelease Notes
|
||||
|
||||
Use the same `RELEASES.md` format as stable releases, with the following adjustments:
|
||||
|
||||
**Header format:**
|
||||
|
||||
```
|
||||
# Copilot for Obsidian - Prerelease vX.Y.Z-<tag>.<N> 🧪
|
||||
```
|
||||
|
||||
The `🧪` emoji signals testing intent. Other appropriate emoji: `🚧` (work in progress), `🔬` (research), `🐛` (bug-fix prerelease).
|
||||
|
||||
**Opening line:** State this is a prerelease intended for testers. Mention what is being tested.
|
||||
|
||||
> Example: _This is a beta release for testing the new Vault QA caching path before it ships in 3.2.9. Please report any indexing or query issues in Discord._
|
||||
|
||||
**Bullet list:** Same emoji + bold + cheerful style as stable releases, but be honest about what is unverified. If a feature has known sharp edges, say so explicitly.
|
||||
|
||||
**Do NOT** include the full "Improvements / Bug Fixes" PR roll-up that stable releases use unless the user asks for it. Prerelease notes should be short and testing-focused.
|
||||
|
||||
**Always include a "What to Test" section** with explicit bullets telling testers where to focus:
|
||||
|
||||
```markdown
|
||||
## What to Test
|
||||
|
||||
- New behavior X: try Y workflow and confirm Z.
|
||||
- Changed behavior W: confirm it still does what it used to do.
|
||||
- Known sharp edges: list anything you suspect is unstable so testers don't waste time reporting it.
|
||||
```
|
||||
|
||||
**Always include a "How to Install" section.** Most users don't know how to install a prerelease.
|
||||
|
||||
```markdown
|
||||
## How to Install the Prerelease
|
||||
|
||||
1. Download `main.js`, `manifest.json`, and `styles.css` from this prerelease's GitHub release page.
|
||||
2. Replace the same three files in your vault's `.obsidian/plugins/copilot/` folder.
|
||||
3. Reload the plugin (Settings → Community Plugins → toggle Copilot off and back on, or restart Obsidian).
|
||||
4. Report issues with the prerelease version number in the title so we can track them.
|
||||
|
||||
To return to the stable release: reinstall the plugin from Obsidian's community-plugin browser.
|
||||
```
|
||||
|
||||
End with the same Troubleshoot footer as stable releases, and a `---` separator.
|
||||
|
||||
### Step 6: Update RELEASES.md
|
||||
|
||||
Prepend the prerelease entry at the top of `RELEASES.md`, right after the `# Release Notes` header line. Keep all existing entries intact.
|
||||
|
||||
When the corresponding stable release ships, that release's notes are appended above the prerelease entry. The prerelease entry stays in the file as a historical record.
|
||||
|
||||
### Step 7: Commit and Create PR
|
||||
|
||||
Stage all changed files. Note that `manifest-beta.json` is what gets touched for prereleases, NOT `manifest.json`:
|
||||
|
||||
```bash
|
||||
git add package.json package-lock.json manifest-beta.json versions.json RELEASES.md
|
||||
```
|
||||
|
||||
If `git status` shows `manifest.json` modified, something went wrong. `version-bump.mjs` should never touch `manifest.json` during a prerelease bump. Stop and tell the user.
|
||||
|
||||
Commit with message: `prerelease: vX.Y.Z-<tag>.<N>`
|
||||
|
||||
Push and create the PR:
|
||||
|
||||
```bash
|
||||
git push -u origin prerelease/vX.Y.Z-<tag>.<N>
|
||||
gh pr create --title "X.Y.Z-<tag>.<N>" --body "$(cat <<'EOF'
|
||||
## Prerelease vX.Y.Z-<tag>.<N>
|
||||
|
||||
[Paste the prerelease notes content here]
|
||||
|
||||
---
|
||||
Generated by the prerelease agent.
|
||||
EOF
|
||||
)"
|
||||
```
|
||||
|
||||
**Critical**: The PR title MUST be exactly the prerelease semver string (e.g., `3.2.9-beta.1`) with no `v` prefix and nothing else. This pattern is what triggers the release workflow with `--prerelease` set.
|
||||
|
||||
### Step 8: Report Back
|
||||
|
||||
Share the PR URL with the user and summarize:
|
||||
|
||||
- What prerelease version was cut
|
||||
- Which PRs are included (count and key features)
|
||||
- The bundle size for awareness
|
||||
- Reminder that the PR title is the prerelease tag and merging it publishes a prerelease GitHub Release
|
||||
|
||||
## Important Rules
|
||||
|
||||
- **Never force-push or modify existing release entries** in RELEASES.md.
|
||||
- **Always start from latest master** — pull before branching.
|
||||
- **The PR title must be a bare prerelease semver string** in the form `X.Y.Z-<tag>.<N>` (e.g., `3.2.9-beta.1`). No `v` prefix, no extra text. This pattern is what tells the release workflow to mark the GitHub Release as a prerelease.
|
||||
- **Use the stable release agent, not this one, for stable releases.** A title like `3.2.9` (no prerelease suffix) goes to the stable agent's flow.
|
||||
- **Read existing RELEASES.md entries** before writing — match the tone and format exactly. Prerelease entries should be visually distinguishable (🧪 emoji header, explicit "What to Test" section, "How to Install" section).
|
||||
- **Be honest about what is unverified.** Prereleases exist to surface bugs, not to oversell stability. If you would not bet your reputation on a feature, say so in the notes.
|
||||
- **Stop on any pre-flight failure.** Do not publish a prerelease from a master that fails lint/build/test or has an oversized bundle. Report and ask, do not paper over.
|
||||
- **Do not silently change `manifest.minAppVersion` or `manifest.isDesktopOnly`** in a prerelease PR. Same rule as stable releases: those changes belong in dedicated PRs.
|
||||
- **Never modify master's `manifest.json` from a prerelease.** It must always reflect the latest stable release. Obsidian's plugin store relies on this. Prerelease metadata goes into `manifest-beta.json` only.
|
||||
- If `npm version` fails or `version-bump.mjs` doesn't run, manually update `manifest-beta.json` and `versions.json` to match the prerelease semver. Do NOT touch `manifest.json`.
|
||||
242
.claude/agents/release.md
Normal file
|
|
@ -0,0 +1,242 @@
|
|||
---
|
||||
name: release
|
||||
description: Use this agent to create a release PR that triggers the automated release workflow. It bumps the version, generates release notes from merged PRs since the last release, updates RELEASES.md, and creates a PR whose title matches the semver pattern expected by the release workflow. Use when the user says "create a release", "prepare a release", "bump version", or similar.
|
||||
model: sonnet
|
||||
color: green
|
||||
---
|
||||
|
||||
You are a release manager for the Copilot for Obsidian plugin. Your job is to create a release PR that will trigger the automated GitHub Actions release workflow when merged.
|
||||
|
||||
## Release Workflow
|
||||
|
||||
The repository has a GitHub Actions workflow that triggers on PR merge to `master` when the PR title matches a semver pattern (e.g., `3.2.4`, `3.3.0`, `4.0.0`). Your job is to:
|
||||
|
||||
1. **Ask the user** whether this is a `patch`, `minor`, or `major` release
|
||||
2. **Bump the version** using `npm version`
|
||||
3. **Generate release notes** from merged PRs since the last release
|
||||
4. **Update RELEASES.md** with the new release entry
|
||||
5. **Create a PR** with the version number as the title
|
||||
|
||||
## Step-by-Step Process
|
||||
|
||||
### Step 0: Pre-flight Sanity Checks
|
||||
|
||||
Before doing any version bumping, validate the repo is releasable. Stop and surface a problem to the user rather than papering over it.
|
||||
|
||||
1. **Confirm clean working tree on master.**
|
||||
|
||||
```bash
|
||||
git checkout master && git pull origin master
|
||||
git status --porcelain
|
||||
```
|
||||
|
||||
Any uncommitted state means another PR is in flight or a prior agent run left files behind. Stop and ask the user to clarify before continuing.
|
||||
|
||||
2. **Run the full project check.**
|
||||
|
||||
```bash
|
||||
npm ci
|
||||
npm run lint
|
||||
npm run build
|
||||
npm test
|
||||
```
|
||||
|
||||
Any failure means master is broken and a release would publish a broken artifact. Stop, report which step failed, and ask the user how to proceed.
|
||||
|
||||
3. **Inspect the built `main.js` bundle size.**
|
||||
|
||||
```bash
|
||||
ls -lh main.js
|
||||
```
|
||||
|
||||
If `main.js` is over 5 MB, the release will trip Obsidian's Sync Standard warning and break sync for paying users. Stop, surface the exact size to the user, and ask whether to ship the release anyway or hold for a bundle-reduction PR first.
|
||||
|
||||
4. **Verify `manifest.json` integrity.**
|
||||
|
||||
```bash
|
||||
node -p "JSON.stringify(require('./manifest.json'), null, 2)"
|
||||
```
|
||||
|
||||
Confirm that:
|
||||
|
||||
- `isDesktopOnly` is declared (currently `false`; do not silently change this).
|
||||
- `minAppVersion` matches the Obsidian APIs the code actually uses. If a commit since the last release introduced a call that needs a newer minimum, the `minAppVersion` bump belongs in its own dedicated PR with its own review window, not bundled inside this release PR. Stop and tell the user.
|
||||
|
||||
5. **Assert that `manifest.json.version` matches the latest stable GitHub Release.**
|
||||
|
||||
Obsidian's community plugin store reads `manifest.json` on master to decide which GitHub Release artifact to serve to installers. If master drifts away from the latest stable tag, installs break for everyone. Catch the drift loudly before doing anything else:
|
||||
|
||||
```bash
|
||||
# Use /releases/latest which returns only the most-recent non-prerelease,
|
||||
# non-draft release in a single call — works regardless of how many
|
||||
# prereleases have accumulated since the last stable.
|
||||
LATEST_STABLE=$(gh api repos/logancyang/obsidian-copilot/releases/latest -q .tag_name)
|
||||
MASTER_VERSION=$(node -p "require('./manifest.json').version")
|
||||
if [ "$LATEST_STABLE" != "$MASTER_VERSION" ]; then
|
||||
echo "DRIFT: master manifest.json.version='$MASTER_VERSION' but latest stable Release='$LATEST_STABLE'. Stop." >&2
|
||||
exit 1
|
||||
fi
|
||||
```
|
||||
|
||||
Stop and tell the user if this fails. Do not "fix" the drift by bumping `manifest.json` inside a release PR — that needs its own dedicated PR.
|
||||
|
||||
6. **Confirm there are merged PRs to release.**
|
||||
|
||||
```bash
|
||||
git describe --tags --abbrev=0
|
||||
git log --oneline $(git describe --tags --abbrev=0)..HEAD | head
|
||||
```
|
||||
|
||||
If the diff is empty, there is nothing to release. Stop and tell the user.
|
||||
|
||||
Only proceed to Step 1 once all six checks pass.
|
||||
|
||||
### Step 1: Determine Release Type
|
||||
|
||||
Ask the user:
|
||||
|
||||
- **Patch** (bug fixes, small improvements)
|
||||
- **Minor** (new features, enhancements)
|
||||
- **Major** (breaking changes, major rewrites)
|
||||
|
||||
### Step 2: Prepare the Branch
|
||||
|
||||
```bash
|
||||
git checkout master
|
||||
git pull origin master
|
||||
```
|
||||
|
||||
Create a release branch:
|
||||
|
||||
```bash
|
||||
git checkout -b release/vX.Y.Z
|
||||
```
|
||||
|
||||
### Step 3: Bump the Version
|
||||
|
||||
Run `npm version [patch|minor|major] --no-git-tag-version` to bump the version in `package.json`. This also triggers `version-bump.mjs` which updates `manifest.json` and `versions.json`.
|
||||
|
||||
**Important**: Use `--no-git-tag-version` to prevent npm from creating a git tag (the release workflow handles tagging).
|
||||
|
||||
After bumping, read the new version from `package.json` to use in subsequent steps.
|
||||
|
||||
### Step 4: Gather and Understand Merged PRs
|
||||
|
||||
Find the last release tag:
|
||||
|
||||
```bash
|
||||
git describe --tags --abbrev=0
|
||||
```
|
||||
|
||||
List all merged PRs since that tag (paginate to avoid missing entries if there are many):
|
||||
|
||||
```bash
|
||||
gh pr list --state merged --base master --search "merged:>YYYY-MM-DD" --json number,title,author,labels --limit 500
|
||||
```
|
||||
|
||||
If the output is exactly 500 entries, there may be more — repeat with an earlier `--search` cutoff or use `--limit 1000` and re-run.
|
||||
|
||||
Use the tag date as the cutoff. You can get it with:
|
||||
|
||||
```bash
|
||||
git log -1 --format=%ai <tag>
|
||||
```
|
||||
|
||||
**Read every PR's description** to understand what each change actually does. Don't rely on PR titles alone — they are often terse or developer-oriented. Fetch each PR's body:
|
||||
|
||||
```bash
|
||||
gh pr view <NUMBER> --json body,title,author,labels
|
||||
```
|
||||
|
||||
Read through all PR descriptions to understand:
|
||||
|
||||
- What user-facing behavior changed
|
||||
- Why the change was made
|
||||
- Any context that helps you write a better release note
|
||||
|
||||
This understanding is critical for writing accurate, user-facing release notes in the next step.
|
||||
|
||||
### Step 5: Generate Release Notes
|
||||
|
||||
Use your understanding of each PR's description and context to write release notes following the established style in `RELEASES.md`. Study the existing entries carefully:
|
||||
|
||||
**Format rules:**
|
||||
|
||||
- Header: `# Copilot for Obsidian - Release vX.Y.Z` followed by emoji (use 🚀 for minor/major, pick something fitting for patches)
|
||||
- Opening line: A 1-2 sentence cheerful summary of the release highlights
|
||||
- Bullet list of changes with emoji prefixes:
|
||||
- Use relevant emoji for each item (🚀 new features, 🛠️ fixes, ⚡ performance, 🎨 UI, 📂 files, 🌐 web, 💡 models, etc.)
|
||||
- **Bold the feature name** at the start of each bullet
|
||||
- Write in plain, cheerful language — no technical jargon
|
||||
- Attribute contributors with `(@username)` at the end of each bullet
|
||||
- For sub-features, use indented bullets with their own emoji
|
||||
- For minor/major releases, include a "More details in the changelog:" section with:
|
||||
- `### Improvements` — list PRs as `- #NUMBER Description @author`
|
||||
- `### Bug Fixes` — list fix PRs as `- #NUMBER Description @author`
|
||||
- End with the Troubleshoot footer:
|
||||
|
||||
```
|
||||
## Troubleshoot
|
||||
|
||||
- If models are missing, navigate to Copilot settings -> Models tab and click "Refresh Built-in Models".
|
||||
- Please report any issue you see in the member channel!
|
||||
```
|
||||
|
||||
- Add `---` separator after the Troubleshoot section
|
||||
|
||||
**Writing style:**
|
||||
|
||||
- Cheerful and enthusiastic, like you're excited to share good news
|
||||
- No developer jargon — explain features from the user's perspective
|
||||
- Use exclamation marks and emoji naturally (don't overdo it)
|
||||
- Highlight what users can DO, not what changed internally
|
||||
- Group related changes together under descriptive bullets
|
||||
|
||||
### Step 6: Update RELEASES.md
|
||||
|
||||
Prepend the new release entry at the top of `RELEASES.md`, right after the `# Release Notes` header line. Keep all existing entries intact.
|
||||
|
||||
### Step 7: Commit and Create PR
|
||||
|
||||
Stage all changed files:
|
||||
|
||||
```bash
|
||||
git add package.json package-lock.json manifest.json versions.json RELEASES.md
|
||||
```
|
||||
|
||||
Commit with message: `release: vX.Y.Z`
|
||||
|
||||
Push and create the PR:
|
||||
|
||||
```bash
|
||||
git push -u origin release/vX.Y.Z
|
||||
gh pr create --title "X.Y.Z" --body "$(cat <<'EOF'
|
||||
## Release vX.Y.Z
|
||||
|
||||
[Paste the release notes content here]
|
||||
|
||||
---
|
||||
Generated by the release agent.
|
||||
EOF
|
||||
)"
|
||||
```
|
||||
|
||||
**Critical**: The PR title MUST be exactly the version number (e.g., `3.2.4`) with no `v` prefix and nothing else. This is what triggers the automated release workflow on merge.
|
||||
|
||||
### Step 8: Report Back
|
||||
|
||||
Share the PR URL with the user and summarize what was included in the release.
|
||||
|
||||
## Important Rules
|
||||
|
||||
- **Never force-push or modify existing release entries** in RELEASES.md
|
||||
- **Always start from latest master** — pull before branching
|
||||
- **The PR title must be a bare stable semver string** (e.g., `3.2.4`, not `v3.2.4` or `Release 3.2.4`). For prereleases, use the prerelease agent instead.
|
||||
- **Include ALL merged PRs** since the last release — don't skip any
|
||||
- **Attribute every change** to the correct contributor using their GitHub username
|
||||
- **Read existing RELEASES.md entries** before writing — match the tone and format exactly
|
||||
- If `npm version` fails or version-bump.mjs doesn't run, manually update `manifest.json` and `versions.json`
|
||||
- **Do not silently change `manifest.minAppVersion` or `manifest.isDesktopOnly`** in a release PR. Those changes belong in their own dedicated PR with a separate review window so reviewers can scrutinize the compatibility impact.
|
||||
- **Surface bundle-size growth in the release notes** if `main.js` grew significantly since the last release. Users notice, and reviewers do too.
|
||||
- **Stop on any pre-flight failure.** Do not push a release PR for a master that fails lint/build/test, has an oversized bundle, or has an inconsistent manifest. Report and ask, do not paper over.
|
||||
- **Stable releases delete `manifest-beta.json` automatically.** `version-bump.mjs` `git rm`s it when bumping to a stable version, on the rationale that the new stable supersedes any in-flight prerelease. This happens in the version-bump commit; nothing extra to do, but be aware that the diff will show the deletion.
|
||||
6
.cursor/rules/coding-rule.mdc
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
---
|
||||
description:
|
||||
globs:
|
||||
alwaysApply: true
|
||||
---
|
||||
Always use functions from logger.ts for logging
|
||||
4
.cursor/rules/test-rule.mdc
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
---
|
||||
alwaysApply: true
|
||||
---
|
||||
Use "npx" command to run test
|
||||
|
|
@ -1,3 +0,0 @@
|
|||
node_modules/
|
||||
|
||||
main.js
|
||||
52
.eslintrc
|
|
@ -1,52 +0,0 @@
|
|||
{
|
||||
"root": true,
|
||||
"parser": "@typescript-eslint/parser",
|
||||
"env": { "node": true },
|
||||
"plugins": ["@typescript-eslint", "tailwindcss"],
|
||||
"extends": [
|
||||
"eslint:recommended",
|
||||
"plugin:@typescript-eslint/recommended",
|
||||
"plugin:react/recommended",
|
||||
"plugin:react-hooks/recommended",
|
||||
"plugin:tailwindcss/recommended",
|
||||
],
|
||||
"parserOptions": {
|
||||
"sourceType": "module",
|
||||
},
|
||||
"rules": {
|
||||
"no-unused-vars": "off",
|
||||
"@typescript-eslint/no-unused-vars": ["error", { "args": "none" }],
|
||||
"@typescript-eslint/ban-ts-comment": "off",
|
||||
"no-prototype-builtins": "off",
|
||||
"@typescript-eslint/no-empty-function": "off",
|
||||
"@typescript-eslint/no-explicit-any": "off",
|
||||
"react/prop-types": "off",
|
||||
"react-hooks/exhaustive-deps": "error",
|
||||
"tailwindcss/classnames-order": "error",
|
||||
"tailwindcss/enforces-negative-arbitrary-values": "error",
|
||||
"tailwindcss/enforces-shorthand": "error",
|
||||
"tailwindcss/migration-from-tailwind-2": "error",
|
||||
"tailwindcss/no-arbitrary-value": "off",
|
||||
"tailwindcss/no-custom-classname": ["error"],
|
||||
"tailwindcss/no-contradicting-classname": "error",
|
||||
},
|
||||
"overrides": [
|
||||
{
|
||||
"files": ["*.json", "*.jsonc", ".eslintrc"],
|
||||
"parser": "jsonc-eslint-parser",
|
||||
"rules": {
|
||||
"jsonc/auto": "error",
|
||||
},
|
||||
},
|
||||
],
|
||||
"settings": {
|
||||
"react": {
|
||||
"version": "detect",
|
||||
},
|
||||
"tailwindcss": {
|
||||
"callees": ["classnames", "clsx", "ctl", "cn", "cva"],
|
||||
"config": "./tailwind.config.js",
|
||||
"cssFiles": ["**/*.css", "!**/node_modules", "!**/.*", "!**/dist", "!**/build"],
|
||||
},
|
||||
},
|
||||
}
|
||||
6
.github/ISSUE_TEMPLATE/bug_report.md
vendored
|
|
@ -7,11 +7,13 @@ assignees: ""
|
|||
---
|
||||
|
||||
- [ ] Disable all other plugins besides Copilot **(required)**
|
||||
- [ ] Screenshot of note + Copilot chat pane + dev console added **(required)**
|
||||
- [ ] Log file generated via "Copilot: Create Log File" command or Settings -> Advanced -> Create Log File **(required)**
|
||||
- [ ] Screenshot of note + Copilot chat pane + dev console added **(optional)**
|
||||
|
||||
Copilot version:
|
||||
Model used:
|
||||
|
||||
(Bug report without the above will be closed)
|
||||
(Bug reports missing the required items above will be closed)
|
||||
|
||||
**Describe how to reproduce**
|
||||
A clear and concise description of what the bug is. Clear steps to reproduce the behavior
|
||||
|
|
|
|||
2
.github/workflows/node.js.yml
vendored
|
|
@ -7,7 +7,7 @@ on:
|
|||
push:
|
||||
branches: ["master", "main"]
|
||||
pull_request:
|
||||
branches: ["master", "main"]
|
||||
# Run on all pull requests regardless of target branch
|
||||
|
||||
jobs:
|
||||
build:
|
||||
|
|
|
|||
216
.github/workflows/release.yml
vendored
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
# Release workflow: triggered when a PR targeting master is merged and its title
|
||||
# is a semver string. Two formats are accepted:
|
||||
# - Stable release: "X.Y.Z" (e.g. "3.2.3")
|
||||
# - Prerelease: "X.Y.Z-<tag>.<N>" (e.g. "3.2.9-beta.1", "3.3.0-rc.0")
|
||||
# Prerelease titles publish a GitHub Release marked as a prerelease.
|
||||
# The PR title becomes the release tag and title; the PR body becomes the release notes.
|
||||
#
|
||||
# Non-semver PR titles (feature PRs, bug fixes, etc.) are silently ignored —
|
||||
# the workflow runs but exits early without creating a release.
|
||||
|
||||
name: Release
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [closed]
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
release:
|
||||
# Only proceed when the PR was actually merged (not just closed/abandoned).
|
||||
if: github.event.pull_request.merged == true
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
permissions:
|
||||
contents: write # required to create GitHub releases and tags
|
||||
attestations: write # required to publish artifact attestations
|
||||
id-token: write # required for the workflow's OIDC token (used to sign attestations)
|
||||
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
steps:
|
||||
# ── Step 1: Validate that the PR title is a semver (stable or prerelease) ──
|
||||
# Exits without error when the title is not semver so non-release PRs
|
||||
# pass silently. Sets outputs.version, outputs.is_release, and
|
||||
# outputs.is_prerelease for downstream steps.
|
||||
- name: Validate semver PR title
|
||||
id: semver
|
||||
env:
|
||||
PR_TITLE: ${{ github.event.pull_request.title }}
|
||||
run: |
|
||||
if echo "$PR_TITLE" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+$'; then
|
||||
echo "PR title '$PR_TITLE' is a stable semver — proceeding with release."
|
||||
echo "version=$PR_TITLE" >> "$GITHUB_OUTPUT"
|
||||
echo "is_release=true" >> "$GITHUB_OUTPUT"
|
||||
echo "is_prerelease=false" >> "$GITHUB_OUTPUT"
|
||||
elif echo "$PR_TITLE" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+-[0-9A-Za-z-]+\.[0-9]+$'; then
|
||||
echo "PR title '$PR_TITLE' is a prerelease semver — proceeding with prerelease."
|
||||
echo "version=$PR_TITLE" >> "$GITHUB_OUTPUT"
|
||||
echo "is_release=true" >> "$GITHUB_OUTPUT"
|
||||
echo "is_prerelease=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "PR title '$PR_TITLE' is not semver (X.Y.Z or X.Y.Z-tag.N) — skipping release."
|
||||
echo "is_release=false" >> "$GITHUB_OUTPUT"
|
||||
echo "is_prerelease=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# ── Step 2: Checkout the merge commit on master ─────────────────────────
|
||||
- name: Checkout
|
||||
if: steps.semver.outputs.is_release == 'true'
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
# Pin to the exact merge commit SHA so concurrent PRs merging to
|
||||
# master don't cause this workflow to build the wrong code.
|
||||
ref: ${{ github.event.pull_request.merge_commit_sha }}
|
||||
|
||||
# ── Step 3: Verify PR title version matches the source-of-truth manifest ──
|
||||
# For stable releases the source of truth is manifest.json (which the
|
||||
# Obsidian plugin store reads). For prereleases the source of truth is
|
||||
# manifest-beta.json (manifest.json must stay pinned at the latest stable
|
||||
# so the plugin store keeps serving installs correctly).
|
||||
#
|
||||
# For prereleases we ALSO assert that manifest.json on the merge commit
|
||||
# still matches the latest stable GitHub Release tag. This defends against
|
||||
# an old-style version-bump or hand edit accidentally re-poisoning master.
|
||||
- name: Verify version matches manifest
|
||||
if: steps.semver.outputs.is_release == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
VERSION: ${{ steps.semver.outputs.version }}
|
||||
IS_PRERELEASE: ${{ steps.semver.outputs.is_prerelease }}
|
||||
run: |
|
||||
if [ "$IS_PRERELEASE" = "true" ]; then
|
||||
MANIFEST_FILE="manifest-beta.json"
|
||||
|
||||
# Guard: master's manifest.json must still equal the latest stable Release tag.
|
||||
# Use /releases/latest which (per GitHub API) returns only the most-recent
|
||||
# non-prerelease, non-draft release in one call — no pagination concerns
|
||||
# even if many prereleases have accumulated since the last stable.
|
||||
LATEST_STABLE=$(gh api "repos/${GITHUB_REPOSITORY}/releases/latest" -q .tag_name 2>/dev/null || true)
|
||||
STABLE_MANIFEST_VERSION=$(node -p "require('./manifest.json').version")
|
||||
if [ -z "$LATEST_STABLE" ]; then
|
||||
echo "Could not determine latest stable GitHub Release tag; aborting to avoid poisoning manifest.json"
|
||||
exit 1
|
||||
fi
|
||||
if [ "$STABLE_MANIFEST_VERSION" != "$LATEST_STABLE" ]; then
|
||||
echo "manifest.json on merge commit ('$STABLE_MANIFEST_VERSION') drifted from latest stable Release ('$LATEST_STABLE')."
|
||||
echo "A prerelease PR must not modify manifest.json. Refusing to publish."
|
||||
exit 1
|
||||
fi
|
||||
echo "manifest.json drift check passed: $STABLE_MANIFEST_VERSION matches latest stable."
|
||||
else
|
||||
MANIFEST_FILE="manifest.json"
|
||||
fi
|
||||
if [ ! -f "$MANIFEST_FILE" ]; then
|
||||
echo "Expected $MANIFEST_FILE to exist for this release type but it does not."
|
||||
exit 1
|
||||
fi
|
||||
MANIFEST_VERSION=$(node -p "require('./$MANIFEST_FILE').version")
|
||||
if [ "$VERSION" != "$MANIFEST_VERSION" ]; then
|
||||
echo "Version mismatch: PR title='$VERSION', $MANIFEST_FILE='$MANIFEST_VERSION'"
|
||||
exit 1
|
||||
fi
|
||||
echo "Version check passed: $VERSION (verified against $MANIFEST_FILE)"
|
||||
|
||||
# ── Step 4: Set up Node.js ───────────────────────────────────────────────
|
||||
- name: Setup Node.js 22.x
|
||||
if: steps.semver.outputs.is_release == 'true'
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: 22.x
|
||||
cache: npm
|
||||
|
||||
# ── Step 5: Install dependencies ────────────────────────────────────────
|
||||
- name: Install dependencies
|
||||
if: steps.semver.outputs.is_release == 'true'
|
||||
run: npm ci
|
||||
|
||||
# ── Step 6: Build the plugin ─────────────────────────────────────────────
|
||||
- name: Build
|
||||
if: steps.semver.outputs.is_release == 'true'
|
||||
run: npm run build
|
||||
|
||||
# ── Step 7: Write PR body to a file (avoids shell injection) ─────────────
|
||||
# Using an environment variable to pass the PR body prevents special
|
||||
# characters (backticks, quotes, dollar signs, etc.) from being
|
||||
# interpreted by the shell.
|
||||
- name: Write release notes to file
|
||||
if: steps.semver.outputs.is_release == 'true'
|
||||
env:
|
||||
PR_BODY: ${{ github.event.pull_request.body }}
|
||||
run: printf '%s' "$PR_BODY" > /tmp/release-notes.md
|
||||
|
||||
# ── Step 8: Check if release already exists (idempotency guard) ─────────
|
||||
- name: Check if release already exists
|
||||
if: steps.semver.outputs.is_release == 'true'
|
||||
id: check_existing
|
||||
env:
|
||||
VERSION: ${{ steps.semver.outputs.version }}
|
||||
run: |
|
||||
if gh release view "$VERSION" > /dev/null 2>&1; then
|
||||
echo "::warning::Release $VERSION already exists — skipping."
|
||||
echo "exists=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "exists=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# ── Step 9: Prepare release-asset manifest ──────────────────────────────
|
||||
# For a prerelease, the manifest.json asset uploaded to the GitHub
|
||||
# Release must carry the prerelease version (so testers who sideload
|
||||
# the assets get a manifest matching what they downloaded). We swap
|
||||
# manifest-beta.json into manifest.json's place IN THE RUNNER only —
|
||||
# this never gets pushed back to master, so the committed manifest.json
|
||||
# stays pinned to the latest stable.
|
||||
- name: Prepare release-asset manifest
|
||||
if: steps.semver.outputs.is_release == 'true' && steps.semver.outputs.is_prerelease == 'true'
|
||||
run: |
|
||||
cp manifest-beta.json manifest.json
|
||||
echo "Release-asset manifest.json (prerelease):"
|
||||
cat manifest.json
|
||||
|
||||
# ── Step 10: Generate build provenance attestation ──────────────────────
|
||||
# Cryptographically signs main.js / manifest.json / styles.css with the
|
||||
# workflow's OIDC identity and publishes the attestation to Sigstore's
|
||||
# public transparency log. Anyone can later verify a downloaded asset
|
||||
# was actually built by this workflow on this commit with:
|
||||
# gh attestation verify main.js --owner logancyang --repo logancyang/obsidian-copilot
|
||||
#
|
||||
# Runs AFTER the prerelease manifest swap so the attested manifest.json
|
||||
# exactly matches the file uploaded to the GitHub Release.
|
||||
- name: Generate artifact attestation
|
||||
if: steps.semver.outputs.is_release == 'true' && steps.check_existing.outputs.exists != 'true'
|
||||
uses: actions/attest-build-provenance@v2
|
||||
with:
|
||||
subject-path: |
|
||||
main.js
|
||||
manifest.json
|
||||
styles.css
|
||||
|
||||
# ── Step 11: Create GitHub Release ───────────────────────────────────────
|
||||
# --target pins the tag to the exact merge commit SHA so concurrent
|
||||
# merges cannot cause the release tag to point at a different commit.
|
||||
# When the PR title is a prerelease semver (X.Y.Z-tag.N), pass --prerelease
|
||||
# so Obsidian's plugin browser does not offer it as a stable update.
|
||||
# Artifacts: main.js, manifest.json, styles.css
|
||||
- name: Create GitHub Release
|
||||
if: steps.semver.outputs.is_release == 'true' && steps.check_existing.outputs.exists != 'true'
|
||||
env:
|
||||
VERSION: ${{ steps.semver.outputs.version }}
|
||||
MERGE_SHA: ${{ github.event.pull_request.merge_commit_sha }}
|
||||
IS_PRERELEASE: ${{ steps.semver.outputs.is_prerelease }}
|
||||
run: |
|
||||
PRERELEASE_FLAG=""
|
||||
if [ "$IS_PRERELEASE" = "true" ]; then
|
||||
PRERELEASE_FLAG="--prerelease"
|
||||
fi
|
||||
gh release create "$VERSION" \
|
||||
--target "$MERGE_SHA" \
|
||||
--title "$VERSION" \
|
||||
--notes-file /tmp/release-notes.md \
|
||||
$PRERELEASE_FLAG \
|
||||
main.js \
|
||||
manifest.json \
|
||||
styles.css
|
||||
4
.gitignore
vendored
|
|
@ -27,3 +27,7 @@ data.json
|
|||
|
||||
# Claude configuration
|
||||
.claude/settings.local.json
|
||||
.claude/worktrees/
|
||||
|
||||
# Development session tracking
|
||||
TODO.md
|
||||
|
|
|
|||
7
.husky/pre-commit
Normal file → Executable file
|
|
@ -1,6 +1 @@
|
|||
# .husky/pre-commit
|
||||
prettier $(git diff --cached --name-only --diff-filter=ACMR | sed 's| |\\ |g') --write --ignore-unknown
|
||||
git update-index --again
|
||||
|
||||
# Add linting
|
||||
npm run lint
|
||||
npx nano-staged
|
||||
|
|
|
|||
391
AGENTS.md
Normal file
|
|
@ -0,0 +1,391 @@
|
|||
# AGENTS.md
|
||||
|
||||
This file provides guidance to any coding agent when working with code in this repository.
|
||||
|
||||
## Overview
|
||||
|
||||
Copilot for Obsidian is an AI-powered assistant plugin that integrates various LLM providers (OpenAI, Anthropic, Google, etc.) with Obsidian. It provides chat interfaces, autocomplete, semantic search, and various AI-powered commands for note-taking and knowledge management.
|
||||
|
||||
## Development Commands
|
||||
|
||||
### Build & Development
|
||||
|
||||
- **NEVER RUN `npm run dev`** - The user will handle all builds manually
|
||||
- `npm run build` - Production build (TypeScript check + minified output)
|
||||
- `npm run test:vault` - macOS only. Installs deps, builds, symlinks `main.js` / `manifest.json` / `styles.css` from the current worktree into `$COPILOT_TEST_VAULT_PATH/.obsidian/plugins/copilot/`, then reloads the plugin via the Obsidian CLI. Requires the user-level env var `COPILOT_TEST_VAULT_PATH` to be set to a vault that has been opened in Obsidian at least once. Use this when the user asks you to load the plugin into their test vault — it replaces manual build + copy + reload.
|
||||
|
||||
### Code Quality
|
||||
|
||||
- `npm run lint` - Run ESLint checks
|
||||
- `npm run lint:fix` - Auto-fix ESLint issues
|
||||
- `npm run format` - Format code with Prettier
|
||||
- `npm run format:check` - Check formatting without changing files
|
||||
- **Before PR:** Always run `npm run format && npm run lint`
|
||||
|
||||
### Testing
|
||||
|
||||
- `npm run test` - Run unit tests (excludes integration tests)
|
||||
- `npm run test:integration` - Run integration tests (requires API keys)
|
||||
- Run single test: `npm test -- -t "test name"`
|
||||
|
||||
### Obsidian CLI (Live Testing)
|
||||
|
||||
The Obsidian desktop app includes a CLI for plugin development. Use the full path:
|
||||
|
||||
```bash
|
||||
/Applications/Obsidian.app/Contents/MacOS/obsidian <command>
|
||||
```
|
||||
|
||||
**Plugin reload** (after `npm run build`):
|
||||
|
||||
```bash
|
||||
/Applications/Obsidian.app/Contents/MacOS/obsidian plugin:reload id=copilot
|
||||
```
|
||||
|
||||
**Console debugging** (requires attaching debugger first):
|
||||
|
||||
```bash
|
||||
/Applications/Obsidian.app/Contents/MacOS/obsidian dev:debug on
|
||||
/Applications/Obsidian.app/Contents/MacOS/obsidian dev:console limit=30
|
||||
/Applications/Obsidian.app/Contents/MacOS/obsidian dev:console level=error limit=10
|
||||
/Applications/Obsidian.app/Contents/MacOS/obsidian dev:errors
|
||||
```
|
||||
|
||||
**Other useful dev commands**:
|
||||
|
||||
- `dev:dom selector=<css>` — Query DOM elements
|
||||
- `dev:screenshot path=<file>` — Take a screenshot
|
||||
- `eval code=<js>` — Execute JS in the app context
|
||||
- `plugin:disable id=copilot` / `plugin:enable id=copilot`
|
||||
|
||||
Run `obsidian help` for the full command list.
|
||||
|
||||
## High-Level Architecture
|
||||
|
||||
### Core Systems
|
||||
|
||||
1. **LLM Provider System** (`src/LLMProviders/`)
|
||||
|
||||
- Provider implementations for OpenAI, Anthropic, Google, Azure, local models
|
||||
- `LLMProviderManager` handles provider lifecycle and switching
|
||||
- Stream-based responses with error handling and rate limiting
|
||||
- Custom model configuration support
|
||||
|
||||
2. **Chain Factory Pattern** (`src/chainFactory.ts`)
|
||||
|
||||
- Different chain types for various AI operations (chat, copilot, adhoc prompts)
|
||||
- LangChain integration for complex workflows
|
||||
- Memory management for conversation context
|
||||
- Tool integration (search, file operations, time queries)
|
||||
|
||||
3. **Vector Store & Search** (`src/search/`)
|
||||
|
||||
- `VectorStoreManager` manages embeddings and semantic search
|
||||
- `ChunkedStorage` for efficient large document handling
|
||||
- Event-driven index updates via `IndexManager`
|
||||
- Multiple embedding providers support
|
||||
|
||||
4. **UI Component System** (`src/components/`)
|
||||
|
||||
- React functional components with Radix UI primitives
|
||||
- Tailwind CSS with class variance authority (CVA)
|
||||
- Modal system for user interactions
|
||||
- Chat interface with streaming support
|
||||
- Settings UI with versioned components
|
||||
|
||||
5. **Message Management Architecture** (`src/core/`, `src/state/`)
|
||||
|
||||
- **MessageRepository** (`src/core/MessageRepository.ts`): Single source of truth for all messages
|
||||
- Stores each message once with both `displayText` and `processedText`
|
||||
- Provides computed views for UI display and LLM processing
|
||||
- No complex dual-array synchronization
|
||||
- **ChatManager** (`src/core/ChatManager.ts`): Central business logic coordinator
|
||||
- Orchestrates MessageRepository, ContextManager, and LLM operations
|
||||
- Handles message sending, editing, regeneration, and deletion
|
||||
- Manages context processing and chain memory synchronization
|
||||
- **Project Chat Isolation**: Maintains separate MessageRepository per project
|
||||
- Automatically detects project switches via `getCurrentMessageRepo()`
|
||||
- Each project has its own isolated message history
|
||||
- Non-project chats use `defaultProjectKey` repository
|
||||
- **ChatUIState** (`src/state/ChatUIState.ts`): Clean UI-only state manager
|
||||
- Delegates all business logic to ChatManager
|
||||
- Provides React integration with subscription mechanism
|
||||
- Replaces legacy SharedState with minimal, focused approach
|
||||
- **ContextManager** (`src/core/ContextManager.ts`): Handles context processing
|
||||
- Processes message context (notes, URLs, selected text)
|
||||
- Reprocesses context when messages are edited
|
||||
|
||||
6. **Settings Management**
|
||||
|
||||
- Jotai for atomic settings state management
|
||||
- React contexts for feature-specific state
|
||||
|
||||
7. **Plugin Integration**
|
||||
- Main entry: `src/main.ts` extends Obsidian Plugin
|
||||
- Command registration system
|
||||
- Event handling for Obsidian lifecycle
|
||||
- Settings persistence and migration
|
||||
- Chat history loading via pending message mechanism
|
||||
|
||||
### Key Patterns
|
||||
|
||||
- **Single Source of Truth**: MessageRepository stores each message once with computed views
|
||||
- **Clean Architecture**: Repository → Manager → UIState → React Components
|
||||
- **Context Reprocessing**: Automatic context updates when messages are edited
|
||||
- **Computed Views**: Display messages for UI, LLM messages for AI processing
|
||||
- **Project Isolation**: Each project maintains its own MessageRepository instance
|
||||
- **Error Handling**: Custom error types with detailed interfaces
|
||||
- **Async Operations**: Consistent async/await pattern with proper error boundaries
|
||||
- **Caching**: Multi-layer caching for files, PDFs, and API responses
|
||||
- **Streaming**: Real-time streaming for LLM responses
|
||||
- **Testing**: Unit tests adjacent to implementation, integration tests for API calls
|
||||
|
||||
## Message Management Architecture
|
||||
|
||||
For detailed architecture diagrams and documentation, see [`MESSAGE_ARCHITECTURE.md`](./designdocs/MESSAGE_ARCHITECTURE.md).
|
||||
|
||||
### Core Classes and Flow
|
||||
|
||||
1. **MessageRepository** (`src/core/MessageRepository.ts`)
|
||||
|
||||
- Single source of truth for all messages
|
||||
- Stores `StoredMessage` objects with both `displayText` and `processedText`
|
||||
- Provides computed views via `getDisplayMessages()` and `getLLMMessages()`
|
||||
- No complex dual-array synchronization or ID matching
|
||||
|
||||
2. **ChatManager** (`src/core/ChatManager.ts`)
|
||||
|
||||
- Central business logic coordinator
|
||||
- Orchestrates MessageRepository, ContextManager, and LLM operations
|
||||
- Handles all message CRUD operations with proper error handling
|
||||
- Synchronizes with chain memory for conversation history
|
||||
- **Project Chat Isolation Implementation**:
|
||||
- Maintains `projectMessageRepos: Map<string, MessageRepository>` for project-specific storage
|
||||
- `getCurrentMessageRepo()` automatically detects current project and returns correct repository
|
||||
- Seamlessly switches between project repositories when project changes
|
||||
- Creates new empty repository for each project (no message caching)
|
||||
|
||||
3. **ChatUIState** (`src/state/ChatUIState.ts`)
|
||||
|
||||
- Clean UI-only state manager
|
||||
- Delegates all business logic to ChatManager
|
||||
- Provides React integration with subscription mechanism
|
||||
- Replaces legacy SharedState with minimal, focused approach
|
||||
|
||||
4. **ContextManager** (`src/core/ContextManager.ts`)
|
||||
|
||||
- Handles context processing (notes, URLs, selected text)
|
||||
- Reprocesses context when messages are edited
|
||||
- Ensures fresh context for LLM processing
|
||||
|
||||
5. **ChatPersistenceManager** (`src/core/ChatPersistenceManager.ts`)
|
||||
- Handles saving and loading chat history to/from markdown files
|
||||
- Project-aware file naming (prefixes with project ID)
|
||||
- Parses and formats chat content for storage
|
||||
- Integrated with ChatManager for seamless persistence
|
||||
|
||||
## Code Style Guidelines
|
||||
|
||||
### MAJOR PRINCIPLES
|
||||
|
||||
- **ALWAYS WRITE GENERALIZABLE SOLUTIONS**: Never add edge-case handling or hardcoded logic for specific scenarios (like "piano notes" or "daily notes"). Solutions must work for all cases.
|
||||
- **NEVER MODIFY AI PROMPT CONTENT**: Do not update, edit, or change any AI prompts, system prompts, or model adapter prompts unless explicitly asked to do so by the user
|
||||
- **Avoid hardcoding**: No hardcoded folder names, file patterns, or special-case logic
|
||||
- **Configuration over convention**: If behavior needs to vary, make it configurable, not hardcoded
|
||||
- **Universal patterns**: Solutions should work equally well for any folder structure, naming convention, or content type
|
||||
|
||||
### TypeScript
|
||||
|
||||
- Strict mode enabled (no implicit any, strict null checks)
|
||||
- Use absolute imports with `@/` prefix: `import { ChainType } from "@/chainFactory"`
|
||||
- Prefer const assertions and type inference where appropriate
|
||||
- Use interface for object shapes, type for unions/aliases
|
||||
|
||||
### React
|
||||
|
||||
- Functional components only (no class components)
|
||||
- Custom hooks for reusable logic
|
||||
- Props interfaces defined above components
|
||||
- Avoid inline styles, use Tailwind classes
|
||||
|
||||
### General
|
||||
|
||||
- File naming: PascalCase for components, camelCase for utilities
|
||||
- Async/await over promises
|
||||
- Early returns for error conditions
|
||||
- **Always add JSDoc comments** for all functions and methods
|
||||
- Organize imports: React → external → internal
|
||||
- **Avoid language-specific lists** (like stopwords or action verbs) - use language-agnostic approaches instead
|
||||
|
||||
### Logging
|
||||
|
||||
- **NEVER use console.log** - Use the logging utilities instead:
|
||||
- `logInfo()` for informational messages
|
||||
- `logWarn()` for warnings
|
||||
- `logError()` for errors
|
||||
- Import from logger: `import { logInfo, logWarn, logError } from "@/logger"`
|
||||
|
||||
### CSS & Styling
|
||||
|
||||
- **NEVER edit `styles.css` directly** - This is a generated file
|
||||
- **Source file**: `src/styles/tailwind.css` - Edit this file for custom CSS
|
||||
- **Build process**: `npm run build:tailwind` compiles `src/styles/tailwind.css` → `styles.css`
|
||||
- **Tailwind classes**: Use Tailwind utility classes in components (see `tailwind.config.js` for available classes)
|
||||
- **Custom CSS**: Add custom styles to `src/styles/tailwind.css` after the `@import` statements
|
||||
- After editing CSS, always run `npm run build` to regenerate `styles.css`
|
||||
|
||||
## Testing Guidelines
|
||||
|
||||
- Unit tests use Jest with TypeScript support
|
||||
- Mock Obsidian API for plugin testing
|
||||
- Integration tests require API keys in `.env.test`
|
||||
- Test files adjacent to implementation (`.test.ts`)
|
||||
- Use `@testing-library/react` for component testing
|
||||
|
||||
### Avoiding Deep Dependency Chains in Tests
|
||||
|
||||
This codebase has deep transitive import chains (e.g. a utility → cache → searchUtils → embeddingManager → brevilabsClient → plusUtils → Modal). Importing any module in this chain from a test requires mocking the entire tree, which is brittle and verbose.
|
||||
|
||||
**Rules for new code:**
|
||||
|
||||
1. **Pass data, not services** — If a function only needs a string (like `outputFolder`), accept it as a parameter. Don't give it access to the entire settings singleton.
|
||||
2. **Singletons at the edges only** — `getSettings()`, `PDFCache.getInstance()`, `BrevilabsClient.getInstance()` should only be called in top-level orchestration (constructors, main entry points). Inner functions receive what they need as parameters.
|
||||
3. **Pure logic in leaf modules** — Extract testable logic into small files with minimal imports. The orchestration file (which has heavy imports) calls the leaf function and passes in the dependencies. See `src/tools/convertedDocOutput.ts` as an example.
|
||||
4. **Litmus test before writing a function** — "Can I test this by calling it directly with plain arguments?" If the answer is no because of an import, that dependency should be a parameter instead.
|
||||
|
||||
## Development Session Planning
|
||||
|
||||
### Using TODO.md for Session Management
|
||||
|
||||
**IMPORTANT**: When working on a development session, maintain a comprehensive `TODO.md` file that serves as the central plan and tracker:
|
||||
|
||||
1. **Session Goal**: Define the high-level objective at the start
|
||||
2. **Task Tracking**:
|
||||
- List all completed tasks with [x] checkboxes
|
||||
- Track pending tasks with [ ] checkboxes
|
||||
- Group related tasks into logical sections
|
||||
3. **Architecture Decisions**: Document key design choices and rationale
|
||||
4. **Progress Updates**: Keep the TODO.md updated as tasks complete
|
||||
5. **Testing Checklist**: Include verification steps for the session
|
||||
|
||||
The TODO.md should be:
|
||||
|
||||
- The single source of truth for session progress
|
||||
- Updated frequently as work progresses
|
||||
- Clear enough that another developer can understand what was done
|
||||
- Comprehensive enough to serve as a migration guide
|
||||
|
||||
### Structure Example:
|
||||
|
||||
```markdown
|
||||
# Development Session TODO
|
||||
|
||||
## Session Goal
|
||||
|
||||
[Clear statement of what this session aims to achieve]
|
||||
|
||||
## Completed Tasks ✅
|
||||
|
||||
- [x] Task description with key details
|
||||
- [x] Another completed task
|
||||
|
||||
## Pending Tasks 📋
|
||||
|
||||
- [ ] Next task to work on
|
||||
- [ ] Future enhancement
|
||||
|
||||
## Architecture Summary
|
||||
|
||||
[Key design decisions and rationale]
|
||||
|
||||
## Testing Checklist
|
||||
|
||||
- [ ] Functionality verification
|
||||
- [ ] Performance checks
|
||||
```
|
||||
|
||||
## Important Notes
|
||||
|
||||
- The plugin supports multiple LLM providers with custom endpoints
|
||||
- Vector store requires rebuilding when switching embedding providers
|
||||
- Settings are versioned - migrations may be needed
|
||||
- Local model support available via Ollama/LM Studio
|
||||
- Rate limiting is implemented for all API calls
|
||||
- For technical debt and known issues, see [`TECHDEBT.md`](./designdocs/todo/TECHDEBT.md)
|
||||
- For current development session planning, see [`TODO.md`](./TODO.md)
|
||||
|
||||
## User-Facing Documentation
|
||||
|
||||
- **When modifying user-facing behavior** (new features, changed settings, removed functionality), **update the corresponding doc in `docs/`**. The doc filenames match their topics (e.g., `llm-providers.md` for provider changes, `agent-mode-and-tools.md` for tool changes).
|
||||
- Docs are written for non-technical users — no source code references, explain behavior and concepts.
|
||||
- If a change affects multiple docs, update all of them.
|
||||
- If you're unsure which doc to update, check `docs/index.md` for the full list with descriptions.
|
||||
|
||||
### AWS Bedrock Usage
|
||||
|
||||
**IMPORTANT**: When using AWS Bedrock, always use **cross-region inference profile IDs** for better reliability and availability:
|
||||
|
||||
- **Global** (recommended): `global.anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
- Routes to any commercial AWS region automatically
|
||||
- Best for reliability and performance
|
||||
- **US**: `us.anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
- **EU**: `eu.anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
- **APAC**: `apac.anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
|
||||
❌ **Avoid regional model IDs** (without prefix): `anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
|
||||
- These only work in specific regions and often fail
|
||||
- Not recommended for production use
|
||||
|
||||
**References:**
|
||||
|
||||
- [AWS Bedrock Cross-Region Inference](https://docs.aws.amazon.com/bedrock/latest/userguide/cross-region-inference.html)
|
||||
- [Supported Inference Profiles](https://docs.aws.amazon.com/bedrock/latest/userguide/inference-profiles-support.html)
|
||||
|
||||
### Obsidian Plugin Environment
|
||||
|
||||
- **Global `app` variable**: In Obsidian plugins, `app` is a globally available variable that provides access to the Obsidian API. It's automatically available in all files without needing to import or declare it.
|
||||
|
||||
### Picking the right `document` / `window` (popout-window safety)
|
||||
|
||||
Obsidian supports pop-out windows. The plugin loads in the main window but views can live in any window. Picking the wrong `Document` / `Window` produces stale references, off-screen popovers, listeners on the wrong window, or DOM nodes that never render. Use this decision order:
|
||||
|
||||
1. **`element.doc` / `element.win`** — preferred. Obsidian augments every `Node` with `.doc: Document` and `.win: Window` that always reflect the element's current owner. Use whenever you have any DOM node in scope (a ref, an event target, a component's container, a `Range`'s `startContainer`).
|
||||
- `containerRef.current?.doc.addEventListener(...)`
|
||||
- `range.startContainer.win.innerWidth`
|
||||
- `editor.getRootElement()?.doc`
|
||||
2. **`global activeDocument` / `activeWindow`** — fallback only. These point to whichever window is _focused right now_. Correct semantics for actions that follow user focus (e.g., the AddImageModal file picker; selectionchange registration at plugin load), but wrong when the action belongs to a specific view (a chat in a popout while the user clicks back to the main window).
|
||||
3. **`document` / `window` globals** — almost always wrong. They are aliases for the main window even when the user is interacting with a popout. Avoid in new code. If you find yourself reaching for them, it's a sign the surrounding code should be taking a `Document`/`Window` parameter or deriving from a DOM ref.
|
||||
4. **`element.ownerDocument`** — works (standard DOM), but prefer `.doc` for consistency with the codebase. They return the same `Document` for any mounted `HTMLElement`. `.doc` is shorter and typed non-nullable.
|
||||
|
||||
**Listeners that may outlive a window migration:** capture the `Document` / `Window` at registration and remove on the same one:
|
||||
|
||||
```ts
|
||||
const doc = containerRef.current?.doc;
|
||||
if (!doc) return;
|
||||
doc.addEventListener("keydown", handler);
|
||||
return () => doc.removeEventListener("keydown", handler);
|
||||
```
|
||||
|
||||
Do **not** rely on `activeDocument` at registration _and_ removal — it can shift between the two calls if focus moves.
|
||||
|
||||
**View migrated to a new window:** for a view that owns React or other long-lived renderers, register `this.containerEl.onWindowMigrated((win) => { ... })` in `onOpen`. The callback fires when Obsidian reparents the element into a different window's document. Tear down and rebuild the renderer there so it captures the new window. Save the returned destroy function and call it in `onClose` to avoid leaks. `CopilotView` is the canonical example — it unmounts and recreates the React root on migration so Lexical re-binds to the popout's window.
|
||||
|
||||
**Cross-realm `instanceof`:** popout windows have their own `Element`, `MouseEvent`, etc., so standard `instanceof` checks fail across windows. Use Obsidian's `element.instanceOf(HTMLElement)` and `event.instanceOf(MouseEvent)` when checking type across realms.
|
||||
|
||||
**Tests (jsdom):** `jest.setup.js` polyfills `Node.doc` / `Node.win` so plugin code using these properties works under jsdom. Don't add `instanceof` guards that depend on the Obsidian-augmented globals without considering the test environment.
|
||||
|
||||
### Architecture Migration Notes
|
||||
|
||||
- **SharedState Removed**: The legacy `src/sharedState.ts` has been completely removed
|
||||
- **Clean Architecture**: New architecture follows Repository → Manager → UIState → UI pattern
|
||||
- **Single Source of Truth**: All messages stored once in MessageRepository with computed views
|
||||
- **Context Always Fresh**: Context is reprocessed when messages are edited to ensure accuracy
|
||||
- **Chat History Loading**: Uses pending message mechanism through CopilotView → Chat component props
|
||||
- **Project Chat Isolation**: Each project now has completely isolated chat history
|
||||
- Automatic detection of project switches via `ProjectManager.getCurrentProjectId()`
|
||||
- Separate MessageRepository instances per project ID
|
||||
- Non-project chats stored in default repository
|
||||
- Backwards compatible - loads existing messages from ProjectManager cache
|
||||
- Zero configuration required - works automatically
|
||||
- Check @tailwind.config.js to understand what tailwind css classnames are available
|
||||
163
CLAUDE.md
|
|
@ -1,164 +1,3 @@
|
|||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## Overview
|
||||
|
||||
Copilot for Obsidian is an AI-powered assistant plugin that integrates various LLM providers (OpenAI, Anthropic, Google, etc.) with Obsidian. It provides chat interfaces, autocomplete, semantic search, and various AI-powered commands for note-taking and knowledge management.
|
||||
|
||||
## Development Commands
|
||||
|
||||
### Build & Development
|
||||
|
||||
- `npm run dev` - Start development server with hot reload (runs Tailwind CSS + esbuild in watch mode)
|
||||
- `npm run build` - Production build (TypeScript check + minified output)
|
||||
|
||||
### Code Quality
|
||||
|
||||
- `npm run lint` - Run ESLint checks
|
||||
- `npm run lint:fix` - Auto-fix ESLint issues
|
||||
- `npm run format` - Format code with Prettier
|
||||
- `npm run format:check` - Check formatting without changing files
|
||||
- **Before PR:** Always run `npm run format && npm run lint`
|
||||
|
||||
### Testing
|
||||
|
||||
- `npm run test` - Run unit tests (excludes integration tests)
|
||||
- `npm run test:integration` - Run integration tests (requires API keys)
|
||||
- Run single test: `npm test -- -t "test name"`
|
||||
|
||||
## High-Level Architecture
|
||||
|
||||
### Core Systems
|
||||
|
||||
1. **LLM Provider System** (`src/LLMProviders/`)
|
||||
|
||||
- Provider implementations for OpenAI, Anthropic, Google, Azure, local models
|
||||
- `LLMProviderManager` handles provider lifecycle and switching
|
||||
- Stream-based responses with error handling and rate limiting
|
||||
- Custom model configuration support
|
||||
|
||||
2. **Chain Factory Pattern** (`src/chainFactory.ts`)
|
||||
|
||||
- Different chain types for various AI operations (chat, copilot, adhoc prompts)
|
||||
- LangChain integration for complex workflows
|
||||
- Memory management for conversation context
|
||||
- Tool integration (search, file operations, time queries)
|
||||
|
||||
3. **Vector Store & Search** (`src/search/`)
|
||||
|
||||
- `VectorStoreManager` manages embeddings and semantic search
|
||||
- `ChunkedStorage` for efficient large document handling
|
||||
- Event-driven index updates via `IndexManager`
|
||||
- Multiple embedding providers support
|
||||
|
||||
4. **UI Component System** (`src/components/`)
|
||||
|
||||
- React functional components with Radix UI primitives
|
||||
- Tailwind CSS with class variance authority (CVA)
|
||||
- Modal system for user interactions
|
||||
- Chat interface with streaming support
|
||||
- Settings UI with versioned components
|
||||
|
||||
5. **State Management**
|
||||
|
||||
- Jotai for atomic state management
|
||||
- React contexts for feature-specific state
|
||||
- Shared state utilities in `src/sharedState.ts`
|
||||
|
||||
6. **Plugin Integration**
|
||||
- Main entry: `src/main.ts` extends Obsidian Plugin
|
||||
- Command registration system
|
||||
- Event handling for Obsidian lifecycle
|
||||
- Settings persistence and migration
|
||||
|
||||
### Key Patterns
|
||||
|
||||
- **Error Handling**: Custom error types with detailed interfaces
|
||||
- **Async Operations**: Consistent async/await pattern with proper error boundaries
|
||||
- **Caching**: Multi-layer caching for files, PDFs, and API responses
|
||||
- **Streaming**: Real-time streaming for LLM responses
|
||||
- **Testing**: Unit tests adjacent to implementation, integration tests for API calls
|
||||
|
||||
## Code Style Guidelines
|
||||
|
||||
### TypeScript
|
||||
|
||||
- Strict mode enabled (no implicit any, strict null checks)
|
||||
- Use absolute imports with `@/` prefix: `import { ChainType } from "@/chainFactory"`
|
||||
- Prefer const assertions and type inference where appropriate
|
||||
- Use interface for object shapes, type for unions/aliases
|
||||
|
||||
### React
|
||||
|
||||
- Functional components only (no class components)
|
||||
- Custom hooks for reusable logic
|
||||
- Props interfaces defined above components
|
||||
- Avoid inline styles, use Tailwind classes
|
||||
|
||||
### General
|
||||
|
||||
- File naming: PascalCase for components, camelCase for utilities
|
||||
- Async/await over promises
|
||||
- Early returns for error conditions
|
||||
- JSDoc for complex functions
|
||||
- Organize imports: React → external → internal
|
||||
|
||||
## Testing Guidelines
|
||||
|
||||
- Unit tests use Jest with TypeScript support
|
||||
- Mock Obsidian API for plugin testing
|
||||
- Integration tests require API keys in `.env.test`
|
||||
- Test files adjacent to implementation (`.test.ts`)
|
||||
- Use `@testing-library/react` for component testing
|
||||
|
||||
### Manual Test Checklists
|
||||
|
||||
**Important**: After each significant change, generate a manual test checklist document that includes:
|
||||
|
||||
1. **Overview**: Brief description of what changed
|
||||
2. **Test Scenarios**: Specific test cases with steps and expected results
|
||||
3. **Verification Checklist**: List of items to verify functionality
|
||||
4. **Files Modified**: List of changed files for reference
|
||||
|
||||
Example format:
|
||||
|
||||
```markdown
|
||||
# [Feature] Test Instructions
|
||||
|
||||
## Overview
|
||||
|
||||
Brief description of the feature/fix
|
||||
|
||||
## Test Scenarios
|
||||
|
||||
### 1. Test Case Name
|
||||
|
||||
1. Step one
|
||||
2. Step two
|
||||
3. **Expected Result:**
|
||||
- Expected behavior
|
||||
- UI state changes
|
||||
- Data persistence
|
||||
|
||||
### 2. Another Test Case
|
||||
|
||||
[...]
|
||||
|
||||
## Verification Checklist
|
||||
|
||||
- [ ] Core functionality works
|
||||
- [ ] Edge cases handled
|
||||
- [ ] No regressions
|
||||
- [ ] Performance acceptable
|
||||
```
|
||||
|
||||
This helps ensure thorough testing and provides documentation for QA.
|
||||
|
||||
## Important Notes
|
||||
|
||||
- The plugin supports multiple LLM providers with custom endpoints
|
||||
- Vector store requires rebuilding when switching embedding providers
|
||||
- Settings are versioned - migrations may be needed
|
||||
- Local model support available via Ollama/LM Studio
|
||||
- Rate limiting is implemented for all API calls
|
||||
@AGENTS.md
|
||||
|
|
@ -53,6 +53,63 @@ In the case of Copilot for Obsidian, you will need to:
|
|||
|
||||
Try to be descriptive in your branch names and pull requests. Happy coding!
|
||||
|
||||
#### Fast Iteration with `npm run test:vault` (macOS)
|
||||
|
||||
If you work across multiple worktrees or just want one command to build and load the plugin into a test vault, use `npm run test:vault`. It runs `npm install`, builds, symlinks `main.js` / `manifest.json` / `styles.css` from the worktree into the vault's `.obsidian/plugins/copilot/` folder, and reloads the plugin in Obsidian via its CLI.
|
||||
|
||||
**One-time setup:**
|
||||
|
||||
1. Create or pick a vault dedicated to plugin testing and open it in Obsidian at least once so `.obsidian/` is created.
|
||||
2. Enable community plugins in that vault (Settings → Community plugins → Turn on).
|
||||
3. Set an env var pointing at the vault path. Add this to `~/.zshrc`, `~/.bashrc`, or `~/.config/fish/config.fish`:
|
||||
|
||||
```bash
|
||||
export COPILOT_TEST_VAULT_PATH="$HOME/Obsidian/CopilotTestVault"
|
||||
```
|
||||
|
||||
**Per change:**
|
||||
|
||||
From any worktree, run:
|
||||
|
||||
```bash
|
||||
npm run test:vault
|
||||
```
|
||||
|
||||
The script installs deps, builds the plugin, symlinks the build artifacts into the vault, then calls `plugin:enable` and `plugin:reload` on the Obsidian CLI. If Obsidian isn't running, the symlinks are still in place — start Obsidian and the new build will load.
|
||||
|
||||
Because the script symlinks files (not the worktree root), the vault's plugin `data.json` (settings, chat history) stays vault-local and is preserved across worktrees and rebuilds.
|
||||
|
||||
Requires macOS with Obsidian installed at `/Applications/Obsidian.app`.
|
||||
|
||||
## Commit Signing
|
||||
|
||||
Commits to `master` must be signed and verified by GitHub. The easiest path is SSH signing using your existing SSH key.
|
||||
|
||||
1. Configure git to sign with your SSH key:
|
||||
|
||||
```bash
|
||||
git config --global gpg.format ssh
|
||||
git config --global user.signingkey ~/.ssh/id_ed25519.pub
|
||||
git config --global commit.gpgsign true
|
||||
```
|
||||
|
||||
Replace `id_ed25519.pub` with the path to your own public key if different.
|
||||
|
||||
2. Register the same key as a **Signing Key** on GitHub at https://github.com/settings/ssh/new. Set "Key type" to `Signing Key` (this is separate from an Authentication Key, even if it's the same key).
|
||||
|
||||
3. Confirm your commit email matches a verified email on your GitHub account at https://github.com/settings/emails. Otherwise commits show as Unverified even when signed.
|
||||
|
||||
4. Verify locally and on GitHub:
|
||||
|
||||
```bash
|
||||
git commit --allow-empty -m "test signing"
|
||||
git log --show-signature -1
|
||||
```
|
||||
|
||||
After pushing, the commit on github.com should display a green **Verified** badge.
|
||||
|
||||
If you already use GPG, set `gpg.format openpgp` instead and register the GPG public key at https://github.com/settings/gpg/new. Commits merged via the GitHub web UI are auto-signed by GitHub and don't need this setup.
|
||||
|
||||
## Prompt Testing
|
||||
|
||||
If you are making prompt changes, make sure to run the integration tests using the following steps:
|
||||
|
|
|
|||
290
README.md
|
|
@ -22,185 +22,233 @@ The Ultimate AI Assistant for Your Second Brain
|
|||
</a>
|
||||
</p>
|
||||
|
||||
Copilot for Obsidian is your best in‑vault AI assistant, designed to listen, act at the speed of thought, and keep you creating in flow—all within Obsidian’s integrated, tab‑free workspace.
|
||||
## The What
|
||||
|
||||
- **🔒 Your data is 100% yours**: Local storage, no ads, and full control of your API keys.
|
||||
- **🧠 Elevate your second brain**: Tap any OpenAI-compatible or local model to uncover insights, spark connections, and create powerful content.
|
||||
- **🌐 Instant multimedia understanding**: Drop in webpages, YouTube videos, images, PDFs, or real-time web search for quick insights and summaries.
|
||||
- **✍️ Create at the speed of thought**: Launch Prompt Palette or edit with AI in one click—your ideas, amplified effortlessly.
|
||||
_Copilot for Obsidian_ is your in‑vault AI assistant with chat-based vault search, web and YouTube support, powerful context processing, and ever-expanding agentic capabilities within Obsidian's highly customizable workspace - all while keeping your data under **your** control.
|
||||
|
||||
## The Why
|
||||
|
||||
Today's AI giants want **you trapped**: your data on their servers, prompts locked to their models, and switching costs that keep you paying. When they change pricing, shut down features, or terminate your account, you lose everything you built.
|
||||
|
||||
We are building the opposite. Our goal is to create a portable agentic experience with no provider lock-in. **Data is always yours.** Use whatever LLM you like. Imagine that a brand new model drops, you run it on your own hardware, and it already knows about you (_long-term memory_), knows how to run _the same commands and tools_ you have defined over time (as just markdown files), and becomes the thought partner and assistant that you _own_. This is AI that grows with you, not a subscription you're hostage to.
|
||||
|
||||
This is the future we believe in. If you share this vision, please support this project!
|
||||
|
||||
## Key Features
|
||||
|
||||
- **🔒 Your data is 100% yours**: Local search and storage, and full control of your data if you use self-hosted models.
|
||||
- **🧠 Bring Your Own Model**: Tap any OpenAI-compatible or local model to uncover insights, spark connections, and create content.
|
||||
- **🖼️ Multimedia understanding**: Drop in webpages, YouTube videos, images, PDFs, EPUBS, or real-time web search for quick insights.
|
||||
- **🔍 Smart Vault Search**: Search your vault with chat, no setup required. Embeddings are optional. Copilot delivers results right away.
|
||||
- **✍️ Composer and Quick Commands**: Interact with your writing with chat, apply changes with 1 click.
|
||||
- **🗂️ Project Mode**: Create AI-ready context based on folders and tags. Think NotebookLM but inside your vault!
|
||||
- **🤖 Agent Mode (Plus)**: Unlock an autonomous agent with built-in tool calling. No commands needed. Copilot automatically triggers vault, web searches or any other relevant tool when relevant.
|
||||
|
||||
<p align="center">
|
||||
<em>Your AI assistant in Obsidian—powerful yet intuitive, keeping you in the creative flow.</em>
|
||||
<em>Copilot's Agent can call the proper tools on its own upon your request.</em>
|
||||
</p>
|
||||
<p align="center">
|
||||
<img src="./images/product-UI-screenshot.png" alt="Product UI screenshot" width="800"/>
|
||||
<img src="./images/product-ui-screenshot.png" alt="Product UI screenshot" width="800"/>
|
||||
</p>
|
||||
|
||||
## Table of Contents
|
||||
|
||||
- [The What](#the-what)
|
||||
- [The Why](#the-why)
|
||||
- [Key Features](#key-features)
|
||||
- [Copilot v4: Agent Mode, Reimagined 🚀](#copilot-v4-agent-mode-reimagined-)
|
||||
- [Why People Love It ❤️](#why-people-love-it-️)
|
||||
- [Get Started](#get-started)
|
||||
- [Install Obsidian Copilot](#install-obsidian-copilot)
|
||||
- [Set API Keys](#set-api-keys)
|
||||
- [Usage](#usage)
|
||||
- [Free User](#free-user)
|
||||
- [**Chat Mode: reference notes and discuss ideas with Copilot**](#chat-mode-reference-notes-and-discuss-ideas-with-copilot)
|
||||
- [**Vault QA Mode: chat with your entire vault**](#vault-qa-mode-chat-with-your-entire-vault)
|
||||
- [Copilot's Command Palette](#copilots-command-palette)
|
||||
- [**Relevant Notes: notes suggestions based on semantic similarity and links**](#relevant-notes-notes-suggestions-based-on-semantic-similarity-and-links)
|
||||
- [Copilot Plus/Believer](#copilot-plusbeliever)
|
||||
- [**Get Precision Insights From a Specific Time Window**](#get-precision-insights-from-a-specific-time-window)
|
||||
- [**Agent Mode: Autonomous Tool Calling**](#agent-mode-autonomous-tool-calling)
|
||||
- [**Understand Images in Your Notes**](#understand-images-in-your-notes)
|
||||
- [**One Prompt, Every Source—Instant Summaries from PDFs, Videos, and Web**](#one-prompt-every-sourceinstant-summaries-from-pdfs-videos-and-web)
|
||||
- [**Need Help?**](#need-help)
|
||||
- [**FAQ**](#faq)
|
||||
- [**🙏 Thank You**](#-thank-you)
|
||||
- [**Copilot Plus Disclosure**](#copilot-plus-disclosure)
|
||||
- [**Authors**](#authors)
|
||||
|
||||
## Copilot v4: Agent Mode, Reimagined 🚀
|
||||
|
||||
Our biggest leap yet. **Copilot v4** lets you run the most capable coding agents available — **opencode**, **Claude Code**, or **Codex** — natively inside your vault, tuned for knowledge work and entirely on your terms. Bring your own agent, keep every note on your device, and let it plan, search, and act across your Second Brain. No lock-in, no compromise.
|
||||
|
||||
**Join Supporter to experience the magic of Copilot v4 now!**
|
||||
|
||||
👉 **[Discover Copilot v4 →](https://www.obsidiancopilot.com/v4)**
|
||||
|
||||
## Why People Love It ❤️
|
||||
|
||||
- *"Copilot is the missing link that turns Obsidian into a true second brain. I use it to draft investment memos with text, code, and visuals—all in one place. It’s the first tool that truly unifies how I search, process, organize, and retrieve knowledge without ever leaving Obsidian. With AI-powered search, organization, and reasoning built into my notes, it unlocks insights I’d otherwise miss. My workflow is faster, deeper, and more connected than ever—I can’t imagine working without it."* - @jasonzhangb, Investor & Research Analyst
|
||||
- *"Since discovering Copilot, my writing process has been completely transformed. Conversing with my own articles and thoughts is the most refreshing experience I’ve had in decades.”* - Mat QV, Writer
|
||||
- *"Copilot has transformed our family—not just as a productivity assistant, but as a therapist. I introduced it to my non‑technical wife, Mania, who was stressed about our daughter’s upcoming exam; within an hour, she gained clarity on her mindset and next steps, finding calm and confidence."* - @screenfluent, A Loving Husband
|
||||
|
||||
## **Get Started in 5 Minutes**
|
||||
## Get Started
|
||||
|
||||
### FREE Product Features
|
||||
### Install Obsidian Copilot
|
||||
|
||||
**🔌 Install Copilot in Community Plugins in Obsidian**
|
||||
1. Open **Obsidian → Settings → Community plugins**.
|
||||
2. Turn off **Safe mode** (if enabled).
|
||||
3. Click **Browse**, search for **“Copilot for Obsidian”**.
|
||||
4. Click **Install**, then **Enable**.
|
||||
|
||||
**🔑 Set Up Your AI Model (API Key)**
|
||||
### Set API Keys
|
||||
|
||||
- To start using Copilot AI features, you'll need access to an AI model of your choice.
|
||||
**Free User**
|
||||
|
||||
1. Go to **Obsidian → Settings → Copilot → Basic** and click **Set Keys**.
|
||||
2. Choose your AI provider(s) (e.g., **OpenRouter, Gemini, OpenAI, Anthropic, Cohere**) and paste your API key(s). **OpenRouter is recommended.**
|
||||
|
||||
**Copilot Plus/Believer**
|
||||
|
||||
1. Copy your license key at your [dashboard](https://www.obsidiancopilot.com/en/dashboard). _Don’t forget to join our wonderful Discord community!_
|
||||
2. Go to **Obsidian → Settings → Copilot → Basic** and paste the key into in the **Copilot Plus** card.
|
||||
|
||||
## Usage
|
||||
|
||||
### Free User
|
||||
|
||||
#### **Chat Mode: reference notes and discuss ideas with Copilot**
|
||||
|
||||
Use `@` to add context and chat with your note.
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=mzMbiamzOqM" target="_blank">
|
||||
<img src="./images/AI-Model-Setup.png" alt="AI Model API Key" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/Add-Context.png" alt="Chat Mode" width="700">
|
||||
</p>
|
||||
|
||||
**📖** **Chat Mode: Summarize Specific Notes**
|
||||
Ask Copilot:
|
||||
|
||||
- 🧠 **Use When:** You want to reference specific notes or folders, generate content, or talk through ideas with Copilot like a knowledgeable thought partner.
|
||||
|
||||
- 💭 **In `Chat` mode, ask Copilot:**
|
||||
> _"Summarize [[Meeting Notes – March]] and create a follow-up task list based on notes in {projects}."_
|
||||
> _Summarize [[Q3 Retrospective]] and identify the top 3 action items for Q4 based on the notes in {01-Projects}._
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=idit7nCqEs0" target="_blank">
|
||||
<img src="./images/Chat-Mode.png" alt="Chat Mode" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/Chat-Mode.png" alt="Chat Mode" width="700">
|
||||
</p>
|
||||
|
||||
**📖** **Vault QA Mode: Chat With Your Entire Vault**
|
||||
#### **Vault QA Mode: chat with your entire vault**
|
||||
|
||||
- 🧠 **Use When:** You want to search your vault for patterns, ideas, or facts without knowing exactly where the information is stored.
|
||||
Ask Copilot:
|
||||
|
||||
- 💭 **In `Vault QA` mode, ask Copilot:**
|
||||
|
||||
> _"What insights can I gather about the benefits of journaling from all of my notes?"_
|
||||
|
||||
- 💡 **Tip:** Replace _the benefits of journaling_ with any topic mentioned in your notes to get more precise results.
|
||||
> _What are the recurring themes in my research regarding the intersection of AI and SaaS?_
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=hBLMWE8WRFU" target="_blank">
|
||||
<img src="./images/Vault-Mode.png" alt="Vault Mode" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/Vault-Mode.png" alt="Vault Mode" width="700">
|
||||
</p>
|
||||
|
||||
**📖 Edit and Apply with One Click**
|
||||
#### Copilot's Command Palette
|
||||
|
||||
- 🧠 **Use When:** You want to quickly fix grammar, spelling or wording directly in your notes—without switching tabs or manually rewriting.
|
||||
Copilot's Command Palette puts powerful AI capabilities at your fingertips. Access all commands in chat window via `/` or via
|
||||
right-click menu on selected text.
|
||||
|
||||
- 💭 **Select the text** and **edit with one RIGHT click**
|
||||
**Add selection to chat context**
|
||||
|
||||
- 💡 **Tip:** Set up and customize your right-click menu with common actions you use often, like _"Summarize"_, _"Simplify Language"_, or _"Translate to Formal Tone"_—so you can apply them effortlessly while you write.
|
||||
Select text and add it to context. Recommend shortcut: `ctrl/cmd + L`
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=hSmRnmEVoec" target="_blank">
|
||||
<img src="./images/One-Click-Commands.png" alt="One-Click Commands" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/Add-Selection-to-Context.png" alt="Add Selection to Context" width="700">
|
||||
</p>
|
||||
|
||||
**📖 Automate your workflow with the Copilot Prompt Palette**
|
||||
**Quick Command**
|
||||
|
||||
- 🧠 **Use When:** You want to speed up repetitive tasks like summarizing, rewriting, or translating without typing full prompts every time.
|
||||
|
||||
- 💭 Type / to use Prompt Palette
|
||||
|
||||
- 💡 **Tip:** Create shortcuts for your most-used actions—like _"Translate to Spanish"_ or _"Draft a blog post outline"_—and trigger them instantly with typing / !
|
||||
Select text and apply action without opening chat. Recommend shortcut: `ctrl/cmd + K`
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=9YzY2OJ54wM" target="_blank">
|
||||
<img src="./images/Prompt-Palette.png" alt="Prompt Palette" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/Quick-Command.png" alt="Quick Command" width="700">
|
||||
</p>
|
||||
|
||||
**📖 Stay in flow with the Relevant Notes**
|
||||
**Edit and Apply with One Click**
|
||||
|
||||
- 🧠 **Use When:** You're working on a note and want to pull in context or insights from related notes—without breaking your focus.
|
||||
|
||||
- 💭 Appears automatically when there's useful related content.
|
||||
|
||||
- 💡 **Tip:** Use it to quickly reference past research, ideas, or decisions—no need to search or switch tabs.
|
||||
Select text and edit with one RIGHT click.
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=qapQD7jD3Uk" target="_blank">
|
||||
<img src="./images/Relevant-Notes.png" alt="Relevant Notes" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/One-Click-Commands.png" alt="One-Click Commands" width="700">
|
||||
</p>
|
||||
|
||||
### Level Up with Copilot Plus and Beyond
|
||||
**Create your Command**
|
||||
|
||||
Create commands and workflows in `Settings → Copilot → Command → Add Cmd`.
|
||||
|
||||
<p align="center">
|
||||
<img src="./images/Create-Command.png" alt="Create Command" width="700">
|
||||
</p>
|
||||
|
||||
**Command Palette in Chat**
|
||||
|
||||
Type `/` to use Command Palette in chat window.
|
||||
|
||||
<p align="center">
|
||||
<img src="./images/Prompt-Palette.png" alt="Prompt Palette" width="700">
|
||||
</p>
|
||||
|
||||
#### **Relevant Notes: notes suggestions based on semantic similarity and links**
|
||||
|
||||
Appears automatically when there's useful related content and links.
|
||||
|
||||
Use it to quickly reference past research, ideas, or decisions—no need to search or switch tabs.
|
||||
|
||||
<p align="center">
|
||||
<img src="./images/Relevant-Notes.png" alt="Relevant Notes" width="700">
|
||||
</p>
|
||||
|
||||
### Copilot Plus/Believer
|
||||
|
||||
Copilot Plus brings powerful AI agentic capabilities, context-aware actions and seamless tool integration—built to elevate your knowledge work in Obsidian.
|
||||
|
||||
🆙 **Upgrade to Copilot Plus**
|
||||
#### **Get Precision Insights From a Specific Time Window**
|
||||
|
||||
First, go to https://www.obsidiancopilot.com/en to subscribe to Copilot Plus. Then, set up Copilot Plus License Key in Obsidian.
|
||||
In agent mode, ask copilot:
|
||||
|
||||
> _What did I do last week?_
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=pPfWKZnNYhA" target="_blank">
|
||||
<img src="./images/Copilot-Plus-Setup.png" alt="Copilot Plus Setup" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/Time-Based-Queries.png" alt="Time-Based Queries" width="700">
|
||||
</p>
|
||||
|
||||
❔Community is at the heart of everything we build. Join us on Discord for updates, priority support, and a voice in shaping the best AI products for your experience.
|
||||
#### **Agent Mode: Autonomous Tool Calling**
|
||||
|
||||
Copilot's agent automatically calls the right tools—no manual commands needed. Just ask, and it searches the web, queries your vault, and combines insights seamlessly.
|
||||
|
||||
Ask Copilot in agent mode:
|
||||
|
||||
> _Research web and my vault and draft a note on AI SaaS onboarding best practices._
|
||||
|
||||
<p align="center">
|
||||
<img src="./images/discord-support.png" alt="Discord support screenshot" width="700"/>
|
||||
<img src="./images/Agent-Mode.png" alt="Agent Mode" width="700">
|
||||
</p>
|
||||
|
||||
**📖 Get Precision Insights From a Specific Time Window**
|
||||
#### **Understand Images in Your Notes**
|
||||
|
||||
- 🧠 **Use When:** You want to quickly review tasks, notes, or ideas from a specific time range without manually digging through files.
|
||||
Copilot can analyze images embedded in your notes—from wireframes and diagrams to screenshots and photos. Get detailed feedback, suggestions, and insights based on visual content.
|
||||
|
||||
- 💭 **In Chat mode, ask Copilot:**
|
||||
Ask Copilot to analyze your wireframes:
|
||||
|
||||
> _"Give me a recap of everything I captured last week."_
|
||||
|
||||
- 💡 **Tip:** Try variations like _"Summarize my highlights from August 11 through August 22"_ for even more insights.
|
||||
> _Analyze the wireframe in [[UX Design - Mobile App Wireframes]] and suggest improvements for the navigation flow._
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=sXP2sjvrqtI" target="_blank">
|
||||
<img src="./images/Time-Based-Queries.png" alt="Time-Based Queries" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/Note-Image.png" alt="Image Understanding" width="700">
|
||||
</p>
|
||||
|
||||
**📖 One Prompt, Every Source—Instant Summaries from PDFs, Videos, and Web**
|
||||
#### **One Prompt, Every Source—Instant Summaries from PDFs, Videos, and Web**
|
||||
|
||||
- 🧠 **Use When:** You want to combine information from multiple formats—documents, videos, web pages, and images—into one concise, actionable summary.
|
||||
In agent mode, ask Copilot
|
||||
|
||||
- 💭 **In PLUS mode, ask Copilot:**
|
||||
> \*Compare the information about [Agent Memory] from this youtube video: [URL], this PDF [file], and @web[search results]. Start with your
|
||||
|
||||
> "Please write a short intro of Kiwi birds based on the following information I collected about this animal.
|
||||
|
||||
> @youtube Summarize [](https://www.youtube.com/watch?v=tZ2jm_UPc6c&t=417s)[https://www.youtube.com/watch?v=tZ2jm_UPc6c&t=417s](https://www.youtube.com/watch?v=ABTfc5wUT1U)
|
||||
> in a short paragraph.
|
||||
|
||||
> @websearch where can I find Kiwi birds?
|
||||
|
||||
> Summarize https://www.doc.govt.nz/nature/native-animals/birds/birds-a-z/kiwi/ in 300 words.“
|
||||
|
||||
- 🛠️ **Add PDFs and Images as Context to Enrich Your Learning**
|
||||
|
||||
- 💡 _Tip: For large PDFs, reference specific sections to focus the AI's attention._
|
||||
conclusion in bullet points in your response*
|
||||
|
||||
<p align="center">
|
||||
<a href="https://www.youtube.com/watch?v=WXoOZmMSHVE" target="_blank">
|
||||
<img src="./images/One-Prompt-Every-Source.png" alt="One Prompt, Every Source" width="700" height="394">
|
||||
</a><br>
|
||||
<em>Click the image to watch the video on YouTube</em>
|
||||
<img src="./images/One-Prompt-Every-Source.png" alt="One Prompt, Every Source" width="700">
|
||||
</p>
|
||||
|
||||
|
||||
# **💡 Need Help?**
|
||||
## **Need Help?**
|
||||
|
||||
- Check the [documentation](https://www.obsidiancopilot.com/en/docs) for setup guides, how-tos, and advanced features.
|
||||
- Watch [Youtube](https://www.youtube.com/@loganhallucinates) for walkthroughs.
|
||||
|
|
@ -219,7 +267,7 @@ First, go to https://www.obsidiancopilot.com/en to subscribe to Copilot Plus. Th
|
|||
- ☑️Clearly describe the feature, why it matters, and how it would help
|
||||
- ☑️Submit your feature request [here](https://github.com/logancyang/obsidian-copilot/issues/new?template=feature_request.md)
|
||||
|
||||
# **🙋♂️ FAQ**
|
||||
## **FAQ**
|
||||
|
||||
<details>
|
||||
<summary><strong>Why isn’t Vault search finding my notes?</strong></summary>
|
||||
|
|
@ -265,28 +313,6 @@ Please refer to your model provider’s documentation for the context window siz
|
|||
|
||||
</details>
|
||||
|
||||
# **💎 Choose the Copilot Plan That’s Right for You**
|
||||
|
||||
| **Feature** | **Free Plan ✅** | **Plus Plan 💎** | **Believer Plan 🛡️** |
|
||||
| ------------------------------------------------------------------ | ---------------- | ---------------- | -------------------- |
|
||||
| No credit card or sign-up required | ✅ | ❌ | ❌ |
|
||||
| All open-source features | ✅ | ✅ | ✅ |
|
||||
| Bring your own API key | ✅ | ✅ | ✅ |
|
||||
| Best-in-class AI chat in Obsidian | ✅ | ✅ | ✅ |
|
||||
| Local data store for Vault QA | ✅ | ✅ | ✅ |
|
||||
| Support | ✅ Essential | ✅ Pro | ✅ Elite |
|
||||
| AI agent capabilities | ❌ | ✅ | ✅ |
|
||||
| Image and PDF support | ❌ | ✅ | ✅ |
|
||||
| Enhanced chat UI (context menu) | ❌ | ✅ | ✅ |
|
||||
| State-of-the-art embedding models included | ❌ | ✅ | ✅ |
|
||||
| Exclusive @AI tools (e.g., web, YouTube) | ❌ | ✅ | ✅ |
|
||||
| Exclusive chat model included in plan | ❌ | ✅ | ✅ |
|
||||
| Access to exclusive Discord channel | ❌ | ✅ | ✅ |
|
||||
| Lifetime access | ❌ | ❌ | ✅ |
|
||||
| Priority access to new features | ❌ | ❌ | ✅ |
|
||||
| Prioritized feature requests | ❌ | ❌ | ✅ |
|
||||
| Exclusive access to next-gen chat & embedding models (coming soon) | ❌ | ❌ | ✅ |
|
||||
|
||||
## **🙏 Thank You**
|
||||
|
||||
If you share the vision of building the most powerful AI agent for our second brain, consider [sponsoring this project](https://github.com/sponsors/logancyang) or buying me a coffee. Help spread the word by sharing Copilot for Obsidian on Twitter/X, Reddit, or your favorite platform!
|
||||
|
|
@ -304,9 +330,13 @@ Special thanks to our top sponsors: @mikelaaron, @pedramamini, @Arlorean, @dashi
|
|||
Copilot Plus is a premium product of Brevilabs LLC and it is not affiliated with Obsidian. It offers a powerful agentic AI integration into Obsidian. Please check out our website [obsidiancopilot.com](https://obsidiancopilot.com/) for more details!
|
||||
|
||||
- An account and payment are required for full access.
|
||||
- Copilot Plus requires network use to faciliate the AI agent.
|
||||
- Copilot Plus does not access your files without your consent.
|
||||
- Copilot Plus collect server-side telemetry to improve the product. Please see the privacy policy on the website for more details.
|
||||
- Copilot Plus requires network use to facilitate the AI agent.
|
||||
- **Privacy & Data Handling**:
|
||||
- **Free tier**: Your messages and notes are sent only to your configured LLM provider (OpenAI, Anthropic, Google, etc.). Nothing goes to Brevilabs servers.
|
||||
- **Plus tier**: Messages go to your configured LLM provider. File conversions (PDF, DOCX, EPUB, images, etc.) are processed by Brevilabs servers only when you explicitly trigger these features via `@` commands.
|
||||
- **Processing vs. Retention**: We process your data to deliver the feature you requested, then discard it. No message content, file uploads, or documents are retained on our servers after processing.
|
||||
- **User ID**: A randomly generated UUID is sent with Plus API requests for service delivery (license abuse prevention, rate limiting) but is not used for user tracking, profiling, or analytics.
|
||||
- Please see the privacy policy on the website for more details.
|
||||
- The frontend code of Copilot plugin is fully open-source. However, the backend code facilitating the AI agents is close-sourced and proprietary.
|
||||
- We offer a full refund if you are not satisfied with the product within 14 days of your purchase, no questions asked.
|
||||
|
||||
|
|
|
|||
2238
RELEASES.md
Normal file
|
|
@ -1,8 +1,25 @@
|
|||
// __mocks__/obsidian.js
|
||||
/* eslint-disable no-undef */
|
||||
import yaml from "js-yaml";
|
||||
import { parse as parseYamlString } from "yaml";
|
||||
|
||||
// Per-test overrides set via the exported `__setRequestUrlImpl` helper.
|
||||
// Default: empty success response. Tests that exercise network paths should
|
||||
// install their own implementation.
|
||||
let requestUrlImpl = jest.fn().mockResolvedValue({
|
||||
status: 200,
|
||||
text: "",
|
||||
json: undefined,
|
||||
arrayBuffer: new ArrayBuffer(0),
|
||||
headers: {},
|
||||
});
|
||||
|
||||
module.exports = {
|
||||
// Reason: normalizePath is used by projectPaths.ts; identity function is sufficient for tests
|
||||
normalizePath: jest.fn().mockImplementation((p) => p),
|
||||
moment: jest.requireActual("moment"),
|
||||
requestUrl: (...args) => requestUrlImpl(...args),
|
||||
__setRequestUrlImpl: (impl) => {
|
||||
requestUrlImpl = impl;
|
||||
},
|
||||
Vault: jest.fn().mockImplementation(() => {
|
||||
return {
|
||||
getMarkdownFiles: jest.fn().mockImplementation(() => {
|
||||
|
|
@ -30,6 +47,78 @@ module.exports = {
|
|||
isDesktop: true,
|
||||
},
|
||||
parseYaml: jest.fn().mockImplementation((content) => {
|
||||
return yaml.load(content);
|
||||
return parseYamlString(content);
|
||||
}),
|
||||
Modal: class Modal {
|
||||
constructor() {
|
||||
this.open = jest.fn();
|
||||
this.close = jest.fn();
|
||||
this.onOpen = jest.fn();
|
||||
this.onClose = jest.fn();
|
||||
}
|
||||
},
|
||||
App: jest.fn().mockImplementation(() => ({
|
||||
workspace: {
|
||||
getActiveFile: jest.fn(),
|
||||
},
|
||||
vault: {
|
||||
read: jest.fn(),
|
||||
},
|
||||
})),
|
||||
ItemView: jest.fn().mockImplementation(function () {
|
||||
this.containerEl = window.document.createElement("div");
|
||||
this.onOpen = jest.fn();
|
||||
this.onClose = jest.fn();
|
||||
this.getDisplayText = jest.fn().mockReturnValue("Mock View");
|
||||
this.getViewType = jest.fn().mockReturnValue("mock-view");
|
||||
this.getIcon = jest.fn().mockReturnValue("document");
|
||||
}),
|
||||
Notice: jest.fn().mockImplementation(function (message) {
|
||||
this.message = message;
|
||||
this.noticeEl = window.document.createElement("div");
|
||||
this.hide = jest.fn();
|
||||
}),
|
||||
TFile: jest.fn().mockImplementation(function (path) {
|
||||
this.path = path;
|
||||
this.name = path.split("/").pop();
|
||||
this.basename = this.name.replace(/\.[^/.]+$/, "");
|
||||
this.extension = path.split(".").pop();
|
||||
}),
|
||||
TFolder: jest.fn().mockImplementation(function (path) {
|
||||
this.path = path || "";
|
||||
this.name = this.path.split("/").pop() || "";
|
||||
}),
|
||||
WorkspaceLeaf: jest.fn().mockImplementation(function () {
|
||||
this.view = null;
|
||||
this.setViewState = jest.fn();
|
||||
this.detach = jest.fn();
|
||||
this.getViewState = jest.fn().mockReturnValue({});
|
||||
}),
|
||||
};
|
||||
|
||||
// Mock the global app object
|
||||
window.app = {
|
||||
vault: {
|
||||
getAbstractFileByPath: jest.fn().mockReturnValue({
|
||||
name: "test-file.md",
|
||||
path: "test-file.md",
|
||||
}),
|
||||
read: jest.fn().mockResolvedValue("test content"),
|
||||
modify: jest.fn().mockResolvedValue(undefined),
|
||||
getMarkdownFiles: jest.fn().mockReturnValue([]),
|
||||
getAllLoadedFiles: jest.fn().mockReturnValue([]),
|
||||
},
|
||||
workspace: {
|
||||
getActiveFile: jest.fn().mockReturnValue(null),
|
||||
getLeaf: jest.fn().mockReturnValue({
|
||||
openFile: jest.fn().mockResolvedValue(undefined),
|
||||
}),
|
||||
},
|
||||
metadataCache: {
|
||||
getFirstLinkpathDest: jest.fn().mockReturnValue(null),
|
||||
getFileCache: jest.fn().mockReturnValue(null),
|
||||
},
|
||||
fileManager: {
|
||||
trashFile: jest.fn().mockResolvedValue(undefined),
|
||||
},
|
||||
};
|
||||
|
|
|
|||
193
designdocs/BEDROCK_TOOL_CALLING.md
Normal file
|
|
@ -0,0 +1,193 @@
|
|||
# BedrockChatModel Tool Calling Implementation
|
||||
|
||||
## Status: ✅ IMPLEMENTED
|
||||
|
||||
Native tool/function calling support has been added to the custom `BedrockChatModel` class for Agent mode.
|
||||
|
||||
---
|
||||
|
||||
## Summary
|
||||
|
||||
The `BedrockChatModel` now supports LangChain's native tool calling via `model.bindTools(tools)`, enabling it to work with Agent mode just like `ChatOpenAI`, `ChatAnthropic`, and `ChatGoogleGenerativeAI`.
|
||||
|
||||
---
|
||||
|
||||
## Implementation Details
|
||||
|
||||
### 1. `bindTools()` Method
|
||||
|
||||
```typescript
|
||||
bindTools(tools: StructuredToolInterface[]): BedrockChatModel {
|
||||
const bound = Object.create(this) as BedrockChatModel;
|
||||
bound.boundTools = tools;
|
||||
return bound;
|
||||
}
|
||||
```
|
||||
|
||||
Creates a new instance with tools bound, following LangChain's pattern.
|
||||
|
||||
### 2. Tool Format Conversion
|
||||
|
||||
```typescript
|
||||
private convertToolsToClaude(tools: StructuredToolInterface[]): any[] {
|
||||
return tools.map((tool) => {
|
||||
let inputSchema: any = { type: "object", properties: {} };
|
||||
if (tool.schema) {
|
||||
inputSchema = isInteropZodSchema(tool.schema)
|
||||
? toJsonSchema(tool.schema)
|
||||
: tool.schema;
|
||||
}
|
||||
return {
|
||||
name: tool.name,
|
||||
description: tool.description || "",
|
||||
input_schema: inputSchema,
|
||||
};
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
Uses LangChain's `isInteropZodSchema` and `toJsonSchema` for proper schema conversion.
|
||||
|
||||
### 3. Request Body with Tools
|
||||
|
||||
Tools are included in the request payload when bound:
|
||||
|
||||
```typescript
|
||||
if (this.boundTools && this.boundTools.length > 0) {
|
||||
payload.tools = this.convertToolsToClaude(this.boundTools);
|
||||
}
|
||||
```
|
||||
|
||||
### 4. ToolMessage Handling
|
||||
|
||||
`buildRequestBody` handles `ToolMessage` (tool results) as `tool_result` content blocks:
|
||||
|
||||
```typescript
|
||||
if (messageType === "tool") {
|
||||
const toolMessage = message as ToolMessage;
|
||||
conversation.push({
|
||||
role: "user",
|
||||
content: [
|
||||
{
|
||||
type: "tool_result",
|
||||
tool_use_id: toolMessage.tool_call_id,
|
||||
content: toolResultContent,
|
||||
},
|
||||
],
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
### 5. AIMessage with Tool Calls
|
||||
|
||||
`buildRequestBody` handles `AIMessage` with `tool_calls` as `tool_use` content blocks:
|
||||
|
||||
```typescript
|
||||
if (toolCalls && toolCalls.length > 0) {
|
||||
const contentBlocks: ContentBlock[] = [];
|
||||
// Add text if present
|
||||
// Add tool_use blocks for each tool call
|
||||
for (const tc of toolCalls) {
|
||||
contentBlocks.push({
|
||||
type: "tool_use",
|
||||
id: tc.id || `tool_${Date.now()}`,
|
||||
name: tc.name,
|
||||
input: tc.args as Record<string, unknown>,
|
||||
});
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 6. Non-Streaming Tool Call Extraction
|
||||
|
||||
`_generate` extracts tool calls from Claude's response:
|
||||
|
||||
```typescript
|
||||
private extractToolCalls(data: any): any[] | undefined {
|
||||
if (!Array.isArray(data?.content)) return undefined;
|
||||
const toolUseBlocks = data.content.filter(
|
||||
(block: any) => block.type === "tool_use"
|
||||
);
|
||||
if (toolUseBlocks.length === 0) return undefined;
|
||||
return toolUseBlocks.map((block: any) => ({
|
||||
id: block.id,
|
||||
name: block.name,
|
||||
args: block.input || {},
|
||||
type: "tool_call" as const,
|
||||
}));
|
||||
}
|
||||
```
|
||||
|
||||
### 7. Streaming Tool Call Chunks
|
||||
|
||||
`processStreamEvent` emits `tool_call_chunks` for LangChain's concat mechanism:
|
||||
|
||||
```typescript
|
||||
private extractToolCallChunk(event: any): { id?: string; index: number; name?: string; args?: string } | null {
|
||||
// content_block_start with tool_use - initial tool call info
|
||||
if (event.type === "content_block_start" && event.content_block?.type === "tool_use") {
|
||||
return {
|
||||
id: event.content_block.id,
|
||||
index: event.index ?? 0,
|
||||
name: event.content_block.name,
|
||||
args: "",
|
||||
};
|
||||
}
|
||||
// content_block_delta with input_json_delta - partial tool args
|
||||
if (event.type === "content_block_delta" && event.delta?.type === "input_json_delta") {
|
||||
return {
|
||||
index: event.index ?? 0,
|
||||
args: event.delta.partial_json || "",
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
```
|
||||
|
||||
Tool call chunks are emitted as `AIMessageChunk` with `tool_call_chunks`:
|
||||
|
||||
```typescript
|
||||
const toolCallChunk = this.extractToolCallChunk(innerEvent);
|
||||
if (toolCallChunk) {
|
||||
const messageChunk = new AIMessageChunk({
|
||||
content: "",
|
||||
response_metadata: chunkMetadata,
|
||||
tool_call_chunks: [toolCallChunk],
|
||||
});
|
||||
deltaChunks.push(new ChatGenerationChunk({ message: messageChunk, text: "" }));
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Testing
|
||||
|
||||
```typescript
|
||||
const model = new BedrockChatModel({
|
||||
modelId: "us.anthropic.claude-3-5-sonnet-20241022-v2:0",
|
||||
apiKey: "...",
|
||||
endpoint: "...",
|
||||
streamEndpoint: "...",
|
||||
});
|
||||
|
||||
const tools = [
|
||||
{
|
||||
name: "get_weather",
|
||||
description: "Get weather for a location",
|
||||
schema: z.object({ location: z.string() }),
|
||||
},
|
||||
];
|
||||
|
||||
const boundModel = model.bindTools(tools);
|
||||
const response = await boundModel.invoke([new HumanMessage("What's the weather in Tokyo?")]);
|
||||
|
||||
console.log(response.tool_calls); // Should have tool call
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Reference
|
||||
|
||||
- [Claude Tool Use on Bedrock](https://docs.aws.amazon.com/bedrock/latest/userguide/tool-use.html)
|
||||
- [Anthropic Tool Use Guide](https://docs.anthropic.com/en/docs/build-with-claude/tool-use)
|
||||
- [LangChain ChatAnthropic](https://github.com/langchain-ai/langchainjs/tree/main/libs/langchain-anthropic)
|
||||
48
designdocs/CITATION_IMPLEMENTATION.md
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
# Inline Citation System
|
||||
|
||||
This guide explains how inline citations are produced across Copilot Plus, Vault QA, and web search, and how the feature is exercised by automated tests.
|
||||
|
||||
## Feature Toggle & Surface Area
|
||||
|
||||
- `enableInlineCitations` (default `true`) lives in `src/settings/model.ts` and is exposed in the QA settings UI (`src/settings/v2/components/QASettings.tsx`).
|
||||
- The toggle gates prompt instructions, fallback post-processing, and chat rendering. When disabled the system falls back to a collapsible sources list without inline markers.
|
||||
|
||||
## Pipeline Overview
|
||||
|
||||
1. **Retrieval Conditioning**
|
||||
- Both `CopilotPlusChainRunner.prepareLocalSearchResult` and `VaultQAChainRunner` sanitize note content with `sanitizeContentForCitations` to strip stray `[^n]`/`[n]` markers before prompting.
|
||||
- Retrieved notes receive stable `__sourceId` values and are serialized with `formatSearchResultsForLLM`; `deduplicateSources` keeps the highest-scoring entry per path/title.
|
||||
- A compact source catalog is built via `formatSourceCatalog`, and Copilot Plus caches the first 20 entries in `lastCitationSources` for fallback footnotes.
|
||||
2. **Prompt Assembly**
|
||||
- `CITATION_RULES` and `WEB_CITATION_RULES` live in `src/LLMProviders/chainRunner/utils/citationUtils.ts`.
|
||||
- `getCitationInstructions` (Copilot Plus) and `getQACitationInstructionsConditional` (Vault QA) append guidance and a source catalog only when inline citations are enabled.
|
||||
- Web search calls `getWebSearchCitationInstructions` so external sources emit `[title](url)` definitions while vault answers stay on `[[Note]]` links.
|
||||
3. **Response Safeguards**
|
||||
- `addFallbackSources` appends a `#### Sources` block when the model produced inline markers but no definitions. Detection relies on `hasExistingCitations`, which now accepts alternate headings (e.g., `## Sources`, `Sources -`) and `<summary>Sources</summary>` blocks.
|
||||
- Copilot Plus passes structured `lastCitationSources` into the fallback helper; Vault QA derives titles from the retriever output.
|
||||
4. **Chat Rendering**
|
||||
- `src/components/chat-components/ChatSingleMessage.tsx` always pipes assistant messages through `processInlineCitations`.
|
||||
- The helper extracts the trailing sources section, builds a first-mention map with `buildCitationMap`, normalizes references (`normalizeCitations`) so constructs like `[^7][^8]` become `[1][2]`, and converts definitions (`convertFootnoteDefinitions`) into clickable wiki links or Markdown anchors.
|
||||
- Duplicate definitions collapse via `consolidateDuplicateSources` + `updateCitationsForConsolidation`, keeping numbering stable. When the sources block is not footnote formatted or citations are disabled, the renderer falls back to a simple `<details>` list.
|
||||
|
||||
## Testing
|
||||
|
||||
- `src/LLMProviders/chainRunner/utils/citationUtils.test.ts`
|
||||
- Sanitization, catalog formatting, and fallback insertion.
|
||||
- `hasExistingCitations` coverage for markdown headings, plain `Sources` labels, and `<summary>` wrappers.
|
||||
- Regression suites for non-sequential citations, duplicate source consolidation, and consecutive markers (`[^7][^8]`).
|
||||
- `src/LLMProviders/chainRunner/utils/searchResultUtils.test.ts`
|
||||
- Ensures retrieved documents are serialized with stable IDs and filtered for `includeInContext` before prompting.
|
||||
- `src/tools/ToolResultFormatter.test.ts`
|
||||
- Verifies the local search tool emits JSON with the `{ type: "local_search", documents: [...] }` shape expected by the chain runners.
|
||||
|
||||
## Manual QA Checklist
|
||||
|
||||
- Vault QA turn using only local search: confirm inline `[1]` markers and numbered sources render without duplication.
|
||||
- Mixed Copilot Plus turn (local search + another tool): ensure fallback still works if the model omits the sources block.
|
||||
- Web search answer: verify footnote definitions render as `[title](url)` links when citations are enabled.
|
||||
|
||||
## Watchlist
|
||||
|
||||
- `sanitizeContentForCitations` intentionally strips bracketed numbers; keep an eye on domains (math, law) where literal `[1990]` values might be desirable.
|
||||
- Inline citations remain model-dependent. `addFallbackSources` guarantees a sources list, but the UI still reflects whatever inline markers the provider returns.
|
||||
473
designdocs/CONTEXT_ENGINEERING.md
Normal file
|
|
@ -0,0 +1,473 @@
|
|||
# Context Engineering - Layered Prefix System
|
||||
|
||||
## Table of Contents
|
||||
|
||||
1. [Purpose](#purpose)
|
||||
2. [First-Principles Goals](#first-principles-goals)
|
||||
3. [Current Architecture (Verified)](#current-architecture-verified)
|
||||
4. [Example Chat Walkthrough](#example-chat-walkthrough)
|
||||
5. [Chain Runner Envelope Usage](#chain-runner-envelope-usage)
|
||||
6. [Strengths](#strengths)
|
||||
7. [Known Gaps](#known-gaps)
|
||||
8. [Improvement Roadmap](#improvement-roadmap)
|
||||
9. [Testing and Observability](#testing-and-observability)
|
||||
10. [References](#references)
|
||||
|
||||
---
|
||||
|
||||
## Purpose
|
||||
|
||||
The context envelope system is the canonical prompt-construction pipeline for chat turns.
|
||||
|
||||
It exists to guarantee:
|
||||
|
||||
- reproducible prompt assembly,
|
||||
- minimal duplication across L2/L3/L4,
|
||||
- safe compaction behavior on large context,
|
||||
- and cache-friendly request prefixes for major providers.
|
||||
|
||||
This document is an implementation audit and roadmap based on current production code.
|
||||
|
||||
---
|
||||
|
||||
## First-Principles Goals
|
||||
|
||||
### 1. Reproducibility
|
||||
|
||||
For the same turn inputs, envelope construction should be deterministic and byte-stable.
|
||||
|
||||
### 2. Token Efficiency
|
||||
|
||||
Context artifacts should appear once in canonical form (or as references), not duplicated across layers.
|
||||
|
||||
### 3. Prefix Cache Stability
|
||||
|
||||
Stable content should stay in early request tokens (L1/L2) so implicit provider caching has maximal hit rate.
|
||||
|
||||
### 4. Compaction Safety
|
||||
|
||||
Compaction must preserve answerability and recovery affordances, especially for non-recoverable context.
|
||||
|
||||
### 5. Persistence Parity
|
||||
|
||||
Loading chat history should preserve envelope quality (or deterministically reconstruct it), so behavior after load matches in-session behavior.
|
||||
|
||||
---
|
||||
|
||||
## Current Architecture (Verified)
|
||||
|
||||
### Layer Definitions
|
||||
|
||||
| Layer | Current Source | Update Trigger | Stability |
|
||||
| --------------- | ------------------------------------------------------------ | ------------------------------- | --------- |
|
||||
| **L1_SYSTEM** | `ChatManager.getSystemPromptForMessage()` | settings/memory/project changes | High |
|
||||
| **L2_PREVIOUS** | Auto-promoted previous user-turn L3 segments | per user turn | Medium |
|
||||
| **L3_TURN** | Current-turn context artifacts (notes/URLs/tags/folders/etc) | every user turn | Low |
|
||||
| **L4_STRIP** | Deferred in envelope, injected from LangChain memory | every turn | Low |
|
||||
| **L5_USER** | processed user query (templated user text) | every user turn | Lowest |
|
||||
|
||||
### End-to-End Flow
|
||||
|
||||
1. `ChatManager.sendMessage()` creates a user message and resolves L1 system prompt.
|
||||
2. `ContextManager.processMessageContext()`:
|
||||
- builds L2 from previous user messages' stored envelopes,
|
||||
- processes current-turn context artifacts,
|
||||
- optionally compacts large context,
|
||||
- builds `PromptContextEnvelope` via `PromptContextEngine`.
|
||||
3. `MessageRepository.updateProcessedText()` stores both legacy `processedText` and `contextEnvelope`.
|
||||
4. Chain runners require `contextEnvelope`, convert with `LayerToMessagesConverter`, inject L4 from memory, then append tool context into user-side payload.
|
||||
|
||||
### L2/L3 Smart Referencing
|
||||
|
||||
- L2 is now deduplicated by segment ID with last-write-wins content updates and stable first-seen ordering.
|
||||
- L3 segments whose IDs already exist in L2 are rendered as references; new IDs include full content.
|
||||
- Segment parsing is centralized in `parseContextIntoSegments()` using `contextBlockRegistry` tags.
|
||||
|
||||
### Tool Placement Model
|
||||
|
||||
- System message contains only L1 + L2.
|
||||
- Tool outputs remain turn-scoped and are prepended to user-side content (`CiC` ordering).
|
||||
- This keeps cacheable prefix isolated from tool variability.
|
||||
|
||||
### Persistence Behavior
|
||||
|
||||
- Chat markdown persists message text plus context references (`[Context: ...]`), not full envelopes.
|
||||
- On load, messages are restored without `contextEnvelope`.
|
||||
- Regeneration now has lazy reprocessing: if envelope is missing, `ChatManager.regenerateMessage()` reprocesses the target user message before running the chain.
|
||||
- Continuing chat after load still does not automatically reconstruct historical envelopes for prior turns.
|
||||
|
||||
### Compaction Stack
|
||||
|
||||
- **Turn-time compaction** (`ContextCompactor`): map-reduce summarization when total context exceeds threshold.
|
||||
- **L2 carry-forward compaction** (`compactSegmentForL2` + `L2ContextCompactor`): deterministic structure+preview compression for promoted previous context.
|
||||
|
||||
### L4 Memory Behavior
|
||||
|
||||
- L4 (chat history) is injected by chain runners from LangChain `BufferWindowMemory`.
|
||||
- **Only L5 text (bare user message) is saved to memory** — context artifacts are NOT included. `BaseChainRunner.handleResponse()` extracts `l5Text` from the envelope, falling back to `originalMessage` or `message`.
|
||||
- This prevents duplication: context artifacts already live in L2/L3 via the envelope; baking them into L4 would cause triple-inclusion and waste tokens.
|
||||
- Assistant responses are compacted at save time by `ChatHistoryCompactor` (strips tool result XML) before storage. Agent-mode responses save only the final answer, not the full reasoning/tool-call chain.
|
||||
|
||||
---
|
||||
|
||||
## Example Chat Walkthrough
|
||||
|
||||
This shows the concrete layer contents across a 3-turn conversation. The user attaches `project-spec.md` in Turn 1, adds `api-docs.md` in Turn 2, then drops `api-docs.md` in Turn 3.
|
||||
|
||||
### Turn 1: User adds `project-spec.md`
|
||||
|
||||
```
|
||||
L1 (System):
|
||||
[system prompt + user memory + project instructions]
|
||||
|
||||
L2 (Previous Context Library):
|
||||
(empty — first turn, no prior context)
|
||||
|
||||
L3 (Current Turn Context):
|
||||
<note_context>
|
||||
<title>project-spec</title>
|
||||
<path>project-spec.md</path>
|
||||
<content>... full note content ...</content>
|
||||
</note_context>
|
||||
→ Segment ID: "project-spec.md" (NEW — full content included)
|
||||
|
||||
L4 (Chat History):
|
||||
(empty — first turn)
|
||||
|
||||
L5 (User Message):
|
||||
"Summarize this"
|
||||
```
|
||||
|
||||
After Turn 1, `BaseChainRunner.handleResponse()` saves to memory:
|
||||
|
||||
- Input: `"Summarize this"` (displayText only — no context XML)
|
||||
- Output: `"Here is a summary of the project spec..."`
|
||||
|
||||
### Turn 2: User keeps `project-spec.md`, adds `api-docs.md`
|
||||
|
||||
```
|
||||
L1 (System):
|
||||
[system prompt — stable ✅, cache-friendly]
|
||||
|
||||
L2 (Previous Context Library):
|
||||
<prior_context source="project-spec.md" type="note">
|
||||
Structure: project-spec (project-spec.md) | Preview: ...first 200 chars...
|
||||
</prior_context>
|
||||
→ Segment ID: "project-spec.md" (promoted from Turn 1 L3, compacted for L2)
|
||||
|
||||
L3 (Current Turn Context):
|
||||
Context attached to this message:
|
||||
- project-spec.md
|
||||
|
||||
Find them in the Context Library in the system prompt above.
|
||||
|
||||
<note_context>
|
||||
<title>api-docs</title>
|
||||
<path>docs/api-docs.md</path>
|
||||
<content>... full note content ...</content>
|
||||
</note_context>
|
||||
→ "project-spec.md" rendered as REFERENCE (already in L2)
|
||||
→ "docs/api-docs.md" is NEW — full content included
|
||||
|
||||
L4 (Chat History):
|
||||
Human: "Summarize this"
|
||||
AI: "Here is a summary of the project spec..."
|
||||
→ Only displayText — no context XML in L4
|
||||
|
||||
L5 (User Message):
|
||||
"What endpoints does the API support?"
|
||||
```
|
||||
|
||||
### Turn 3: User keeps `project-spec.md` only (drops `api-docs.md`)
|
||||
|
||||
```
|
||||
L1 (System):
|
||||
[system prompt — stable ✅]
|
||||
|
||||
L2 (Previous Context Library):
|
||||
<prior_context source="project-spec.md" type="note">
|
||||
Structure: project-spec (project-spec.md) | Preview: ...first 200 chars...
|
||||
</prior_context>
|
||||
<prior_context source="docs/api-docs.md" type="note">
|
||||
Structure: api-docs (docs/api-docs.md) | Preview: ...first 200 chars...
|
||||
</prior_context>
|
||||
→ Both deduplicated by segment ID. "project-spec.md" retains its
|
||||
first-seen position; "docs/api-docs.md" added after.
|
||||
→ L2 is CUMULATIVE and STABLE — cache hit for the prefix ✅
|
||||
|
||||
L3 (Current Turn Context):
|
||||
Context attached to this message:
|
||||
- project-spec.md
|
||||
|
||||
Find them in the Context Library in the system prompt above.
|
||||
→ "project-spec.md" is a REFERENCE (in L2)
|
||||
→ "docs/api-docs.md" is NOT referenced (user didn't attach it this turn)
|
||||
but it remains in L2 for cache stability and potential follow-up use
|
||||
|
||||
L4 (Chat History):
|
||||
Human: "Summarize this"
|
||||
AI: "Here is a summary of the project spec..."
|
||||
Human: "What endpoints does the API support?"
|
||||
AI: "The API supports the following endpoints..."
|
||||
→ Clean displayText only — no bloat
|
||||
|
||||
L5 (User Message):
|
||||
"Explain the auth flow from the spec"
|
||||
```
|
||||
|
||||
### Key Behaviors Demonstrated
|
||||
|
||||
| Behavior | Where | Example |
|
||||
| ------------------------------- | ----------- | ----------------------------------------------------------------- |
|
||||
| **Per-artifact segment IDs** | L3 parsing | `"project-spec.md"`, `"docs/api-docs.md"` — not generic `"notes"` |
|
||||
| **L2 dedup (last-write-wins)** | L2 build | Same ID across turns → content updated, position preserved |
|
||||
| **Smart referencing** | L3 render | Items in L2 become `- project-spec.md` references |
|
||||
| **L2 cumulative growth** | L2 library | `api-docs.md` stays in L2 even when dropped from L3 |
|
||||
| **L2 carry-forward compaction** | L2 content | Full `<note_context>` → `<prior_context>` with structure+preview |
|
||||
| **L4 displayText only** | Memory save | `"Summarize this"` — no `<note_context>` XML |
|
||||
| **Prefix cache stability** | L1+L2 | L1 stable across turns; L2 grows monotonically, doesn't shrink |
|
||||
|
||||
---
|
||||
|
||||
## Chain Runner Envelope Usage
|
||||
|
||||
All four chain runners use the context envelope for LLM message construction. Each delegates final response handling to `BaseChainRunner.handleResponse()`, which saves only L5 text (expanded user query, no context XML) to L4 memory.
|
||||
|
||||
### Per-Runner Behavior
|
||||
|
||||
| Runner | Envelope Construction | Tool Results | User Message Source |
|
||||
| ------------------------------ | ------------------------------------------------------------------------------------- | ------------------------------------------------------------- | -------------------- |
|
||||
| **LLMChainRunner** | `LayerToMessagesConverter.convert()` → system (L1+L2), user (L3 refs + L5) | None | Envelope only |
|
||||
| **CopilotPlusChainRunner** | Same converter, then `ensureUserQueryLabel` adds `[User query]:` separator | Prepended to user message in CiC order | L5 text via envelope |
|
||||
| **AutonomousAgentChainRunner** | Same converter for initial messages; ReAct loop appends AI + ToolMessages iteratively | Native tool calling — each result is a separate `ToolMessage` | L5 text via envelope |
|
||||
| **VaultQAChainRunner** | Same converter | Retrieval results via hybrid/lexical retriever | Envelope only |
|
||||
|
||||
### CopilotPlus: Single-Shot Tool Flow
|
||||
|
||||
1. Planning phase analyzes L5 text to determine which `@commands` to execute.
|
||||
2. Tool results (localSearch, web fetch, etc.) are formatted and prepended to the user message using CiC ordering: `[tool results] → [L3 references + L5 with User query label]`.
|
||||
3. Single LLM call with the complete message array: `[system (L1+L2)] → [L4 history] → [user (tools + L3 + L5)]`.
|
||||
|
||||
### Autonomous Agent: ReAct Loop Flow
|
||||
|
||||
1. Initial message array built identically to CopilotPlus: `[system (L1+L2+tool guidelines)] → [L4 history] → [user (L3 refs + L5)]`.
|
||||
2. Model responds with native tool calls (e.g., `localSearch`, `readFile`).
|
||||
3. Each tool result becomes a `ToolMessage` appended to the growing messages array.
|
||||
4. `localSearch` results get CiC ordering: the user's question (from L5 `originalUserPrompt`) is appended after the search payload via `ensureCiCOrderingWithQuestion`.
|
||||
5. Loop repeats until model responds without tool calls (final answer).
|
||||
|
||||
### Token Efficiency Audit
|
||||
|
||||
**Verified efficient (no action needed):**
|
||||
|
||||
- L1+L2 prefix is stable and cacheable across turns — tool results never enter the system message.
|
||||
- L3 uses smart references for artifacts already in L2 — no content duplication in the user message.
|
||||
- L4 contains only displayText (via L5 extraction in `handleResponse`) — no context XML leakage.
|
||||
- Chain runners extract L5 text from the envelope for `cleanedUserMessage` and `originalUserPrompt`, never using `processedText` (which contains L2+L3+L5 concatenated).
|
||||
|
||||
**Known inefficiencies (accepted tradeoffs):**
|
||||
|
||||
| Issue | Severity | Tokens Wasted | Rationale |
|
||||
| ------------------------------------------------------------------------- | -------- | --------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| CiC user-question repeated per `localSearch` tool call in agent loop | LOW | ~50 tokens x N searches | Intentional — each `ToolMessage` is independent; the model needs the question for grounding across ReAct iterations |
|
||||
| L2 content may overlap with `localSearch` results | MEDIUM | Variable (L2 has compacted preview, search has relevant chunks) | Structural limitation — search doesn't know about L2. Overlap is partial since L2 is compacted to structure+preview while search returns precision chunks |
|
||||
| Legacy `processedText` stores L2+L3+L5 concatenation in MessageRepository | LOW | 0 (not sent to LLM) | Storage-only waste. Field is unused by envelope-based chain runners but persisted for backward compatibility |
|
||||
|
||||
---
|
||||
|
||||
## Strengths
|
||||
|
||||
1. Envelope-first prompt construction is now consistent across all active chain runners.
|
||||
2. L2 deduplication by artifact ID prevents linear repeated growth for repeated attachments.
|
||||
3. Segment parsing is centralized and registry-driven, avoiding ad-hoc per-chain parsing logic.
|
||||
4. Uniform tool placement removes previous cache-boundary ambiguity.
|
||||
5. Memory-side assistant compaction reduces L4 bloat from tool payloads.
|
||||
6. Regeneration path is now resilient to missing envelopes on loaded history.
|
||||
|
||||
---
|
||||
|
||||
## Known Gaps
|
||||
|
||||
### P0: No Token Budget Enforcement on Full Payload
|
||||
|
||||
- **No compaction mechanism checks the total assembled payload** (L1+L2+L3+L4+L5) against any budget. Each compactor guards only its own subset.
|
||||
- **L1 (project context) is never budgeted or compacted**: In Projects mode, all project files are concatenated verbatim into L1 with no size limit. This is often the largest single layer.
|
||||
- **Compaction threshold is blind to L1**: `ContextManager` checks L2+L3 against `autoCompactThreshold` (or hardcoded `PROJECT_COMPACT_THRESHOLD = 1M`), but L1 size is never subtracted. The threshold acts as if L2+L3 is the entire payload.
|
||||
- **L4 (chat history) has no remaining-budget awareness**: `loadAndAddChatHistory()` loads all history messages regardless of how much budget L1+L2+L3+L5 have already consumed.
|
||||
- **`contextTurns` is a crude count-based proxy**: `BufferWindowMemory.k = contextTurns * 2` limits by message count, not token size, providing no actual overflow protection.
|
||||
- See [TOKEN_BUDGET_ENFORCEMENT.md](./TOKEN_BUDGET_ENFORCEMENT.md) for detailed analysis and fix plan.
|
||||
|
||||
### P0: Persistence Parity Is Still Incomplete
|
||||
|
||||
- Loaded chats do not rehydrate historical envelopes.
|
||||
- Result: follow-up turns after load do not benefit from historical L2 library unless messages are individually reprocessed.
|
||||
|
||||
### P0: Non-Deterministic Fallback Segment IDs
|
||||
|
||||
- `appendParsedSegments()` uses `unparsed-${Date.now()}` when parsing fails.
|
||||
- This breaks deterministic envelope identity and degrades cache behavior in edge cases.
|
||||
|
||||
### P1: Parser Still Depends on Regex Over Rendered XML
|
||||
|
||||
- `parseContextIntoSegments()` is significantly better than earlier local regex logic, but it still parses serialized XML strings rather than typed artifacts.
|
||||
- Malformed/nested edge cases can still create parse misses and fallback behavior.
|
||||
|
||||
### P1: Compaction Semantics Need Stronger Invariants
|
||||
|
||||
- Two compaction modes exist (LLM summarization vs deterministic L2 preview compaction), but explicit invariants on what must remain verbatim are not enforced centrally.
|
||||
- Non-recoverable context (e.g., selected text) needs stricter protection policies under heavy compaction.
|
||||
|
||||
### P1: L2 Mutation Tradeoff Not Formalized
|
||||
|
||||
- Current policy is ID-dedup + content overwrite.
|
||||
- This is token-efficient, but mutable artifacts can invalidate long cached prefixes when content changes.
|
||||
- The system needs an explicit "freshness vs cache stability" policy.
|
||||
|
||||
### P2: Envelope Metadata Is Underused
|
||||
|
||||
- `conversationId` is currently `null`.
|
||||
- Missing stable conversation-level identity weakens observability and future caching strategies.
|
||||
|
||||
### P2: Documentation Drift Risk
|
||||
|
||||
- Prior docs contained stale statements (e.g., fallback-to-processedText behavior, on-load envelope reconstruction).
|
||||
- This doc now reflects current code; future changes should keep this aligned.
|
||||
|
||||
---
|
||||
|
||||
## Improvement Roadmap
|
||||
|
||||
### Phase 1: Correctness and Determinism
|
||||
|
||||
1. Replace timestamp fallback IDs with deterministic IDs:
|
||||
- `unparsed:${sha256(content)}` (plus optional short prefix by source type).
|
||||
2. Add envelope invariants checker (debug + tests):
|
||||
- no duplicate segment IDs inside a layer,
|
||||
- no L3 full-content block when ID exists in L2 unless explicitly marked override,
|
||||
- stable layer ordering and hash consistency.
|
||||
3. Add robust parse-failure telemetry:
|
||||
- count parse misses,
|
||||
- log failing tag/source metadata,
|
||||
- capture hash-only samples in debug mode.
|
||||
|
||||
### Phase 2: Persistence Parity
|
||||
|
||||
1. Add lazy historical envelope reconstruction on first post-load send:
|
||||
- reprocess only prior user messages that have context references and missing envelopes,
|
||||
- skip URL/web-tab refetch by policy where needed.
|
||||
2. Optional long-term path:
|
||||
- persist compact envelope metadata (or typed artifact snapshots) alongside markdown history for deterministic restoration.
|
||||
|
||||
### Phase 3: Compaction Safety and Policy
|
||||
|
||||
1. Define explicit compaction classes:
|
||||
- recoverable artifacts: can be summarized with re-fetch instructions,
|
||||
- non-recoverable artifacts: preserve verbatim or bounded extractive compaction only.
|
||||
2. Add post-compaction validation:
|
||||
- each compacted artifact must retain deterministic source identity and recoverability hints.
|
||||
3. Make compaction strategy configurable by chain type and context source type.
|
||||
|
||||
### Phase 4: Cache Optimization
|
||||
|
||||
1. Split L1 into stable and mutable subsections (for example:
|
||||
- static system contract,
|
||||
- user memory and project overlays) to reduce unnecessary prefix invalidation.
|
||||
2. Introduce provider-aware cache hooks (opt-in):
|
||||
- Anthropic `cache_control`,
|
||||
- Gemini explicit cache primitives,
|
||||
- keep model-agnostic baseline unchanged.
|
||||
3. Add per-turn prefix hash diff reporting:
|
||||
- `L1 hash`, `L2 hash`, combined prefix hash,
|
||||
- classify why prefix changed (settings, context attach, file change, memory update).
|
||||
|
||||
### Phase 5: Typed Artifact Pipeline (Strategic)
|
||||
|
||||
Move from "render XML then parse XML" to a typed artifact graph:
|
||||
|
||||
- `ContextProcessor` emits typed artifacts directly (`artifactKey`, `sourceType`, `recoverable`, `payload`, `contentHash`).
|
||||
- Envelope stores typed segments as canonical source-of-truth.
|
||||
- XML remains a rendering format, not parsing substrate.
|
||||
|
||||
This is the highest-leverage change for long-term reproducibility and parser robustness.
|
||||
|
||||
### Phase 6: Context Envelope Integration Test Suite
|
||||
|
||||
Build a comprehensive test suite that validates multi-turn envelope behavior without requiring manual UI testing:
|
||||
|
||||
1. **Multi-turn envelope simulation tests**:
|
||||
|
||||
- Simulate 3+ turn conversations with various artifact combinations (notes, URLs, YouTube, PDFs, selected text).
|
||||
- Assert correct L2 promotion, dedup, smart referencing, and compaction at each turn.
|
||||
- Validate that L4 memory contains only displayText (no context XML leakage).
|
||||
|
||||
2. **Layer composition snapshot tests**:
|
||||
|
||||
- For canonical conversation trajectories, snapshot the full `[L1, L2, L3, L4, L5]` payload sent to the LLM.
|
||||
- Detect unintended regressions in layer ordering, dedup behavior, or content placement.
|
||||
|
||||
3. **Round-trip persistence tests**:
|
||||
|
||||
- Save a conversation to markdown, reload it, send a follow-up turn.
|
||||
- Assert that lazy reprocessing reconstructs envelopes and L2 correctly.
|
||||
|
||||
4. **Edge-case regression tests**:
|
||||
|
||||
- Same artifact attached across 5+ turns (dedup stability).
|
||||
- Artifact added, removed, re-added (L2 cumulative behavior).
|
||||
- Multiple `selected_text` blocks in same turn (unique ID generation).
|
||||
- Malformed XML blocks (graceful fallback, no silent data loss).
|
||||
- Very large context triggering compaction (invariants preserved).
|
||||
|
||||
5. **Property-based tests** (optional, aspirational):
|
||||
- Generate random artifact sequences and assert envelope invariants hold:
|
||||
no duplicate segment IDs within a layer, L3 references only exist if ID is in L2,
|
||||
L4 never contains XML block tags.
|
||||
|
||||
This suite replaces the need for manual multi-turn chat testing in the UI and provides a safety net for all future envelope changes.
|
||||
|
||||
---
|
||||
|
||||
## Testing and Observability
|
||||
|
||||
### Core Tests to Add/Strengthen
|
||||
|
||||
1. Post-load follow-up turn should rebuild/rehydrate envelope behavior deterministically.
|
||||
2. Deterministic fallback ID behavior (no wall-clock dependence).
|
||||
3. Property tests for parser with malformed/nested blocks.
|
||||
4. Compaction invariants:
|
||||
- non-recoverable blocks never become unrecoverable summaries without explicit guardrails.
|
||||
5. Prefix-hash stability tests across common conversation trajectories.
|
||||
|
||||
### Runtime Metrics (Debug Mode)
|
||||
|
||||
- Envelope build time by phase (L2 build, context processing, compaction, render).
|
||||
- Segment counts per layer and dedup ratio.
|
||||
- Prefix hash change reason classification.
|
||||
- Parse-failure count and compacted-context proportion.
|
||||
|
||||
---
|
||||
|
||||
## References
|
||||
|
||||
### Primary Implementation Files
|
||||
|
||||
- `src/core/ChatManager.ts`
|
||||
- `src/core/ContextManager.ts`
|
||||
- `src/context/PromptContextTypes.ts`
|
||||
- `src/context/PromptContextEngine.ts`
|
||||
- `src/context/parseContextSegments.ts`
|
||||
- `src/context/LayerToMessagesConverter.ts`
|
||||
- `src/core/MessageRepository.ts`
|
||||
- `src/core/ChatPersistenceManager.ts`
|
||||
- `src/LLMProviders/chainRunner/LLMChainRunner.ts`
|
||||
- `src/LLMProviders/chainRunner/VaultQAChainRunner.ts`
|
||||
- `src/LLMProviders/chainRunner/CopilotPlusChainRunner.ts`
|
||||
- `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts`
|
||||
|
||||
### Related Docs
|
||||
|
||||
- `designdocs/MESSAGE_ARCHITECTURE.md`
|
||||
- `designdocs/TOOLS.md`
|
||||
- `designdocs/NATIVE_TOOL_CALLING_MIGRATION.md`
|
||||
- `designdocs/todo/TECHDEBT.md`
|
||||
- `TODO.md`
|
||||
755
designdocs/MESSAGE_ARCHITECTURE.md
Normal file
|
|
@ -0,0 +1,755 @@
|
|||
# Message Architecture & Context Design
|
||||
|
||||
This document describes the new message management and context processing architecture that replaced the legacy SharedState system. The new design follows clean architecture principles with a single source of truth, computed views, and complete project isolation.
|
||||
|
||||
**Note**: For detailed information about the layered context system (L1-L5 layers, L2 auto-promotion, cache optimization), see [CONTEXT_ENGINEERING.md](./CONTEXT_ENGINEERING.md).
|
||||
|
||||
## Architecture Principles
|
||||
|
||||
### Single Source of Truth
|
||||
|
||||
- Each message is stored exactly once in `MessageRepository`
|
||||
- All UI and LLM views are computed from this single storage
|
||||
- No complex dual-array synchronization or ID matching
|
||||
|
||||
### Clean Architecture Flow
|
||||
|
||||
```
|
||||
User Input → ChatUIState → ChatManager → getCurrentMessageRepo() → MessageRepository + ContextManager
|
||||
↓ ↓
|
||||
ChatPersistenceManager Project-specific storage
|
||||
↓
|
||||
UI Components ← ChatUIState ← Computed Views ← MessageRepository
|
||||
↓
|
||||
LLM Processing ← Chain Memory ← getLLMMessages() ← MessageRepository
|
||||
```
|
||||
|
||||
### Context Always Fresh
|
||||
|
||||
- Context is reprocessed when messages are edited
|
||||
- No stale context issues from cached processing
|
||||
- Ensures accurate context for LLM interactions
|
||||
|
||||
### Project Isolation
|
||||
|
||||
- Each project maintains its own isolated chat history
|
||||
- Automatic detection and switching when project changes
|
||||
- Zero configuration required - works automatically
|
||||
- Non-project chats use a default repository
|
||||
|
||||
## Core Components
|
||||
|
||||
### 1. MessageRepository (`src/core/MessageRepository.ts`)
|
||||
|
||||
**Purpose**: Single source of truth for all messages
|
||||
|
||||
**Key Concepts**:
|
||||
|
||||
- Stores `StoredMessage` objects with both `displayText` and `processedText`
|
||||
- `displayText`: What the user typed or AI responded (for UI display)
|
||||
- `processedText`: For user messages, includes context. For AI messages, same as display
|
||||
|
||||
**Core Methods**:
|
||||
|
||||
```typescript
|
||||
// Add new message
|
||||
addMessage(
|
||||
displayText: string,
|
||||
processedText: string,
|
||||
sender: string,
|
||||
context?: MessageContext,
|
||||
content?: MessageContent[]
|
||||
): string
|
||||
|
||||
// Get computed views
|
||||
getDisplayMessages(): ChatMessage[] // For UI rendering
|
||||
getLLMMessages(): ChatMessage[] // For AI processing
|
||||
|
||||
// Edit operations
|
||||
editMessage(id: string, newDisplayText: string): boolean
|
||||
updateProcessedText(
|
||||
id: string,
|
||||
processedText: string,
|
||||
contextEnvelope?: PromptContextEnvelope
|
||||
): boolean
|
||||
|
||||
// Bulk operations
|
||||
truncateAfterMessageId(messageId: string): void
|
||||
loadMessages(messages: ChatMessage[]): void
|
||||
```
|
||||
|
||||
> **Storing envelopes**: When `ChatManager` calls `addMessage` with a full `NewChatMessage`, it includes the `contextEnvelope` property so the repository keeps both the legacy `processedText` and the canonical layered representation. The string-based overload remains for low-level utilities and test fixtures.
|
||||
|
||||
### 2. ChatManager (`src/core/ChatManager.ts`)
|
||||
|
||||
**Purpose**: Central business logic coordinator
|
||||
|
||||
**Responsibilities**:
|
||||
|
||||
- Orchestrates MessageRepository, ContextManager, and LLM operations
|
||||
- Handles all message CRUD operations with proper error handling
|
||||
- Synchronizes with chain memory for conversation history
|
||||
- Manages context processing lifecycle
|
||||
- **Project Isolation**: Maintains separate MessageRepository per project
|
||||
- **Persistence**: Integrates with ChatPersistenceManager for saving/loading
|
||||
|
||||
**Envelope Lifecycle**:
|
||||
|
||||
1. User sends a message → `ContextManager.processMessageContext()` returns both `processedContent` **and** a `PromptContextEnvelope`.
|
||||
2. `ChatManager` stores the envelope with `MessageRepository.updateProcessedText(...)`.
|
||||
3. When any chain runner executes, it reads `userMessage.contextEnvelope` and feeds it to `LayerToMessagesConverter.convert()` to materialize the L1-L5 prompt structure.
|
||||
4. Regeneration or edits call `reprocessMessageContext`, ensuring a fresh envelope replaces the stale one.
|
||||
|
||||
**Key Operations**:
|
||||
|
||||
```typescript
|
||||
// Send new message with context processing
|
||||
async sendMessage(displayText: string, context: MessageContext, chainType: ChainType, includeActiveNote?: boolean): Promise<string>
|
||||
|
||||
// Edit message and reprocess context
|
||||
async editMessage(messageId: string, newText: string, chainType: ChainType, includeActiveNote?: boolean): Promise<boolean>
|
||||
|
||||
// Regenerate AI response
|
||||
async regenerateMessage(messageId: string, onUpdateMessage: Function, onAddMessage: Function): Promise<boolean>
|
||||
|
||||
// Memory synchronization
|
||||
private async updateChainMemory(): Promise<void>
|
||||
|
||||
// Project management
|
||||
private getCurrentMessageRepo(): MessageRepository // Auto-detects current project
|
||||
async handleProjectSwitch(): Promise<void> // Forces project detection
|
||||
|
||||
// Persistence
|
||||
async saveChat(modelKey: string): Promise<{ success: boolean; path?: string; error?: string }>
|
||||
```
|
||||
|
||||
**Project Isolation Implementation**:
|
||||
|
||||
```typescript
|
||||
// Internal structure
|
||||
private projectMessageRepos: Map<string, MessageRepository>
|
||||
|
||||
// Automatic project detection
|
||||
getCurrentMessageRepo() {
|
||||
const currentProjectId = ProjectManager.getCurrentProjectId() || defaultProjectKey;
|
||||
if (!this.projectMessageRepos.has(currentProjectId)) {
|
||||
// Create new repository for this project
|
||||
const repo = new MessageRepository();
|
||||
this.projectMessageRepos.set(currentProjectId, repo);
|
||||
}
|
||||
return this.projectMessageRepos.get(currentProjectId)!;
|
||||
}
|
||||
```
|
||||
|
||||
### 3. ChatUIState (`src/state/ChatUIState.ts`)
|
||||
|
||||
**Purpose**: Clean UI-only state manager
|
||||
|
||||
**Design Philosophy**:
|
||||
|
||||
- Delegates ALL business logic to ChatManager
|
||||
- Provides React integration with subscription mechanism
|
||||
- Replaces legacy SharedState with minimal, focused approach
|
||||
|
||||
**React Integration**:
|
||||
|
||||
```typescript
|
||||
// Subscribe to state changes
|
||||
subscribe(listener: () => void): () => void
|
||||
|
||||
// Delegate operations to ChatManager
|
||||
async sendMessage(displayText: string, context: MessageContext, chainType: ChainType, includeActiveNote?: boolean): Promise<string>
|
||||
getMessages(): ChatMessage[] // Computed view for UI
|
||||
|
||||
// Project and persistence operations
|
||||
async handleProjectSwitch(): Promise<void> // Handle UI updates for project switch
|
||||
async saveChat(modelKey: string): Promise<{ success: boolean; path?: string; error?: string }>
|
||||
|
||||
// Legacy compatibility (for backward compatibility)
|
||||
get chatHistory(): ChatMessage[]
|
||||
addMessage(message: ChatMessage): void
|
||||
clearChatHistory(): void
|
||||
|
||||
// Notify React components of changes
|
||||
private notifyListeners(): void
|
||||
```
|
||||
|
||||
### 4. ContextManager (`src/core/ContextManager.ts`)
|
||||
|
||||
**Purpose**: Handles context processing and reprocessing with layered context architecture
|
||||
|
||||
**Key Features**:
|
||||
|
||||
- **Layered Context System**: Builds structured L1-L5 context layers (see CONTEXT_ENGINEERING.md)
|
||||
- **L2 Auto-Promotion**: Automatically promotes previous turn's context to L2 for cache stability
|
||||
- **Deduplication**: Ensures context appears only once (L3 takes priority over L2)
|
||||
- **Context Processing**: Handles notes, URLs, selected text, tags, and folders
|
||||
- **Reprocessing**: Regenerates fresh context when messages are edited
|
||||
- **Envelope Building**: Creates `PromptContextEnvelope` with structured layers and hashes
|
||||
- **Chain-Aware Processing**: Applies chain-specific rules (e.g., Copilot Plus URL processing, active-note handling for vision models)
|
||||
|
||||
**Core Methods**:
|
||||
|
||||
```typescript
|
||||
// Process context for new message (includes L2 building from history)
|
||||
async processMessageContext(
|
||||
message: ChatMessage,
|
||||
fileParserManager: FileParserManager,
|
||||
vault: Vault,
|
||||
chainType: ChainType,
|
||||
includeActiveNote: boolean,
|
||||
activeNote: TFile | null,
|
||||
messageRepo: MessageRepository,
|
||||
systemPrompt?: string
|
||||
): Promise<ContextProcessingResult>
|
||||
|
||||
// Reprocess context for edited message
|
||||
async reprocessMessageContext(messageId: string, ...): Promise<void>
|
||||
```
|
||||
|
||||
### 5. ChatPersistenceManager (`src/core/ChatPersistenceManager.ts`)
|
||||
|
||||
**Purpose**: Handles saving and loading chat history to/from markdown files
|
||||
|
||||
**Key Features**:
|
||||
|
||||
- Project-aware file naming (prefixes with project ID)
|
||||
- Filters chat history files based on current project
|
||||
- Parses and formats chat content for storage
|
||||
- Integrated with ChatManager for seamless persistence
|
||||
|
||||
**Core Methods**:
|
||||
|
||||
```typescript
|
||||
// Save chat to markdown file
|
||||
async saveChat(messages: ChatMessage[], modelKey: string, projectId?: string): Promise<{ success: boolean; path?: string; error?: string }>
|
||||
|
||||
// Get available chat history files
|
||||
async getChatHistoryFiles(): Promise<TFile[]>
|
||||
|
||||
// File naming convention
|
||||
// Project chats: `[projectId]-[timestamp]-[modelKey]-chat.md`
|
||||
// Non-project chats: `[timestamp]-[modelKey]-chat.md`
|
||||
```
|
||||
|
||||
## Architecture Diagrams
|
||||
|
||||
### Complete System Architecture
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────────────────────────────┐
|
||||
│ User Interface Layer │
|
||||
├─────────────────────────────────────────────────────────────────────────────────────┤
|
||||
│ │
|
||||
│ ┌─────────────────┐ ┌──────────────────┐ │
|
||||
│ │ Chat.tsx │ ◄────── uses ──────────► │ CopilotView.tsx │ │
|
||||
│ │ │ │ │ │
|
||||
│ └────────┬────────┘ └──────────────────┘ │
|
||||
│ │ │
|
||||
│ │ subscribes to & calls │
|
||||
│ ▼ │
|
||||
└───────────┬─────────────────────────────────────────────────────────────────────────┘
|
||||
│
|
||||
┌───────────┴─────────────────────────────────────────────────────────────────────────┐
|
||||
│ State Layer │
|
||||
├─────────────────────────────────────────────────────────────────────────────────────┤
|
||||
│ │ │
|
||||
│ ┌────────▼────────┐ │
|
||||
│ │ ChatUIState │ - React state management │
|
||||
│ │ │ - Subscription mechanism for UI updates │
|
||||
│ │ │ - Delegates all business logic to ChatManager │
|
||||
│ └────────┬────────┘ │
|
||||
│ │ │
|
||||
└───────────┴─────────────────────────────────────────────────────────────────────────┘
|
||||
│ delegates to
|
||||
┌───────────▼─────────────────────────────────────────────────────────────────────────┐
|
||||
│ Business Logic Layer │
|
||||
├─────────────────────────────────────────────────────────────────────────────────────┤
|
||||
│ │
|
||||
│ ┌─────────────────┐ orchestrates ┌─────────────────────────────┐ │
|
||||
│ │ ChatManager │ ◄──────────────────────────► │ ContextManager (singleton) │ │
|
||||
│ │ │ │ │ │
|
||||
│ │ - Message CRUD │ │ - Process message context │ │
|
||||
│ │ - Project │ │ - Handle note attachments │ │
|
||||
│ │ isolation │ │ - Reprocess on edit │ │
|
||||
│ │ - Memory sync │ └─────────────────────────────┘ │
|
||||
│ └────────┬────────┘ │
|
||||
│ │ │
|
||||
│ │ manages ┌─────────────────────────────┐ │
|
||||
│ │ │ ChatPersistenceManager │ │
|
||||
│ ├──────────────────────────────────────►│ │ │
|
||||
│ │ │ - Save/load chat history │ │
|
||||
│ │ │ - Project-aware file naming │ │
|
||||
│ │ └─────────────────────────────┘ │
|
||||
│ │ │
|
||||
│ │ coordinates ┌─────────────────────────────┐ │
|
||||
│ ├──────────────────────────────────────►│ ChainManager │ │
|
||||
│ │ │ │ │
|
||||
│ │ │ - Memory management │ │
|
||||
│ │ │ - LLM chain operations │ │
|
||||
│ │ └──────────┬──────────────────┘ │
|
||||
│ │ │ │
|
||||
│ │ ▼ │
|
||||
│ │ ┌─────────────────────────────┐ │
|
||||
│ │ │ MemoryManager │ │
|
||||
│ │ │ │ │
|
||||
│ │ │ - Chain memory storage │ │
|
||||
│ │ │ - Conversation history │ │
|
||||
│ │ └─────────────────────────────┘ │
|
||||
└───────────┴─────────────────────────────────────────────────────────────────────────┘
|
||||
│
|
||||
┌───────────▼─────────────────────────────────────────────────────────────────────────┐
|
||||
│ Data Storage Layer │
|
||||
├─────────────────────────────────────────────────────────────────────────────────────┤
|
||||
│ │
|
||||
│ ┌─────────────────────────────────────────────────────────────────────────────┐ │
|
||||
│ │ MessageRepository │ │
|
||||
│ │ │ │
|
||||
│ │ ┌─────────────────┐ Computed Views ┌────────────────────────────┐ │ │
|
||||
│ │ │ StoredMessage[] │ ──────────────────────► │ getDisplayMessages() │ │ │
|
||||
│ │ │ │ │ (for UI rendering) │ │ │
|
||||
│ │ │ - id │ └────────────────────────────┘ │ │
|
||||
│ │ │ - displayText │ │ │
|
||||
│ │ │ - processedText │ ──────────────────────► ┌────────────────────────────┐ │ │
|
||||
│ │ │ - sender │ │ getLLMMessages() │ │ │
|
||||
│ │ │ - timestamp │ │ (for AI processing) │ │ │
|
||||
│ │ │ - context │ └────────────────────────────┘ │ │
|
||||
│ │ └─────────────────┘ │ │
|
||||
│ │ │ │
|
||||
│ │ Single source of truth - no dual storage! │ │
|
||||
│ └─────────────────────────────────────────────────────────────────────────────┘ │
|
||||
│ │
|
||||
└──────────────────────────────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
## Project Isolation Architecture
|
||||
|
||||
### Multi-Repository Design
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────────────────────────────┐
|
||||
│ ChatManager │
|
||||
├─────────────────────────────────────────────────────────────────────────────────────┤
|
||||
│ │
|
||||
│ projectMessageRepos: Map<string, MessageRepository> │
|
||||
│ │
|
||||
│ ┌──────────────────┐ ┌──────────────────┐ ┌──────────────────┐ │
|
||||
│ │ "defaultProject" │ │ "project-1" │ │ "project-2" │ │
|
||||
│ │ │ │ │ │ │ │
|
||||
│ │ MessageRepo │ │ MessageRepo │ │ MessageRepo │ │
|
||||
│ │ - Non-project │ │ - Project 1 │ │ - Project 2 │ │
|
||||
│ │ messages │ │ messages only │ │ messages only │ │
|
||||
│ └──────────────────┘ └──────────────────┘ └──────────────────┘ │
|
||||
│ ▲ ▲ ▲ │
|
||||
│ │ │ │ │
|
||||
│ └─────────────────────────┴─────────────────────────┘ │
|
||||
│ │ │
|
||||
│ getCurrentMessageRepo() │
|
||||
│ (auto-detects active project) │
|
||||
│ │
|
||||
└──────────────────────────────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
### Project Switch Flow
|
||||
|
||||
```
|
||||
Project Switch Detected (via Obsidian workspace)
|
||||
↓
|
||||
ProjectManager.getCurrentProjectId() returns new ID
|
||||
↓
|
||||
ChatManager.getCurrentMessageRepo()
|
||||
↓
|
||||
Check if repository exists for project
|
||||
↓ (if not)
|
||||
Create new MessageRepository
|
||||
↓
|
||||
Store in projectMessageRepos Map
|
||||
↓
|
||||
Return project-specific repository
|
||||
```
|
||||
|
||||
## Message Lifecycle
|
||||
|
||||
### Example: User Message with Context Note
|
||||
|
||||
When a user types "Summarize this note" and attaches "meeting-notes.md":
|
||||
|
||||
1. **Input**: User text + attached file → Chat component
|
||||
2. **Storage**: MessageRepository stores displayText: "Summarize this note"
|
||||
3. **Processing**: ContextManager reads the note and creates processedText with proper XML structure
|
||||
4. **Memory Sync**: Chain memory receives the processed version for LLM
|
||||
5. **UI Update**: Shows message with context badge, displays only "Summarize this note"
|
||||
6. **LLM Processing**: AI receives full context and generates response
|
||||
|
||||
### Context XML Format
|
||||
|
||||
All context is wrapped in semantic XML tags for clear structure:
|
||||
|
||||
> Layered Prompting: During Phase 3 the raw XML is still captured in `processedText` for backward compatibility, but the canonical representation sent to the LLM comes from the envelope layers:
|
||||
>
|
||||
> - **L1_SYSTEM / L2_PREVIOUS**: Stable prefixes rendered from accumulated context
|
||||
> - **L3_TURN**: Turn-specific smart references that either link back to L2 or embed full content
|
||||
> - **L4_STRIP**: Chat history managed by memory
|
||||
> - **L5_USER**: The user’s raw message (plus composer directives when present)
|
||||
|
||||
#### Note Context
|
||||
|
||||
```xml
|
||||
<note_context>
|
||||
<title>meeting-notes</title>
|
||||
<path>docs/meeting-notes.md</path>
|
||||
<ctime>2024-01-15T10:00:00.000Z</ctime>
|
||||
<mtime>2024-01-15T14:30:00.000Z</mtime>
|
||||
<content>
|
||||
[actual note content here]
|
||||
</content>
|
||||
</note_context>
|
||||
```
|
||||
|
||||
#### URL Context
|
||||
|
||||
```xml
|
||||
<url_content>
|
||||
<url>https://example.com/article</url>
|
||||
<content>
|
||||
[fetched content from URL]
|
||||
</content>
|
||||
</url_content>
|
||||
```
|
||||
|
||||
#### Selected Text Context
|
||||
|
||||
```xml
|
||||
<selected_text>
|
||||
<title>Source Note Title</title>
|
||||
<path>path/to/source.md</path>
|
||||
<start_line>45</start_line>
|
||||
<end_line>52</end_line>
|
||||
<content>
|
||||
[selected text content]
|
||||
</content>
|
||||
</selected_text>
|
||||
```
|
||||
|
||||
#### Error Cases
|
||||
|
||||
```xml
|
||||
<note_context_error>
|
||||
<title>filename</title>
|
||||
<path>path/to/file.ext</path>
|
||||
<error>[Error: Could not process file]</error>
|
||||
</note_context_error>
|
||||
```
|
||||
|
||||
This separation ensures:
|
||||
|
||||
- Clean UI (shows what user typed)
|
||||
- Rich context for AI (includes note content)
|
||||
- Reprocessable context on message edits
|
||||
|
||||
For detailed examples, see:
|
||||
|
||||
- `src/core/MessageLifecycle.test.ts` - Complete lifecycle demonstration with context notes
|
||||
- `src/core/MessageLifecycle.xmltags.test.ts` - XML tag formatting tests and examples
|
||||
|
||||
### 1. Sending a New Message
|
||||
|
||||
```
|
||||
User Input
|
||||
↓ (via Chat component)
|
||||
ChatUIState.sendMessage()
|
||||
↓
|
||||
ChatManager.sendMessage()
|
||||
↓
|
||||
MessageRepository.addMessage() // Store with basic content
|
||||
↓
|
||||
ContextManager.processMessageContext() // Add context
|
||||
↓
|
||||
MessageRepository.updateProcessedText() // Update with processed text + context envelope
|
||||
↓
|
||||
ChatManager.updateChainMemory() // Sync to LLM
|
||||
↓
|
||||
ChatUIState.notifyListeners() // Update UI
|
||||
```
|
||||
|
||||
### 2. Editing a Message
|
||||
|
||||
```
|
||||
User Edit
|
||||
↓
|
||||
ChatUIState.editMessage()
|
||||
↓
|
||||
ChatManager.editMessage()
|
||||
↓
|
||||
MessageRepository.editMessage() // Update display text
|
||||
↓
|
||||
ContextManager.reprocessMessageContext() // Fresh context
|
||||
↓
|
||||
ChatManager.updateChainMemory() // Sync to LLM
|
||||
↓
|
||||
ChatUIState.notifyListeners() // Update UI
|
||||
```
|
||||
|
||||
### 3. Message Display
|
||||
|
||||
```
|
||||
React Component Render
|
||||
↓
|
||||
ChatUIState.getMessages()
|
||||
↓
|
||||
ChatManager.getDisplayMessages()
|
||||
↓
|
||||
ChatManager.getCurrentMessageRepo() // Project-aware
|
||||
↓
|
||||
MessageRepository.getDisplayMessages() // Computed view
|
||||
↓
|
||||
Filter visible messages → Map to ChatMessage format
|
||||
```
|
||||
|
||||
### 4. Saving Chat History
|
||||
|
||||
```
|
||||
User Save Action
|
||||
↓
|
||||
Chat.tsx → ChatUIState.saveChat(modelKey)
|
||||
↓
|
||||
ChatManager.saveChat(modelKey)
|
||||
↓
|
||||
Get current project ID and messages
|
||||
↓
|
||||
ChatPersistenceManager.saveChat(messages, modelKey, projectId)
|
||||
↓
|
||||
Create markdown file with project prefix
|
||||
↓
|
||||
Return success with file path
|
||||
```
|
||||
|
||||
### 5. Project Switch
|
||||
|
||||
```
|
||||
Project Change in Obsidian
|
||||
↓
|
||||
ChatUIState.handleProjectSwitch()
|
||||
↓
|
||||
ChatManager.handleProjectSwitch()
|
||||
↓
|
||||
Force getCurrentMessageRepo() to re-detect project
|
||||
↓
|
||||
Switch to different MessageRepository
|
||||
↓
|
||||
Update chain memory with new project's messages
|
||||
↓
|
||||
Notify UI listeners for refresh
|
||||
```
|
||||
|
||||
## Data Structures
|
||||
|
||||
### StoredMessage (Internal)
|
||||
|
||||
```typescript
|
||||
interface StoredMessage {
|
||||
id: string;
|
||||
displayText: string; // What user typed/AI responded
|
||||
processedText: string; // With context for user, same as display for AI
|
||||
sender: string;
|
||||
timestamp: FormattedDateTime;
|
||||
context?: MessageContext;
|
||||
isVisible: boolean;
|
||||
isErrorMessage?: boolean;
|
||||
sources?: { title: string; score: number }[];
|
||||
content?: any[];
|
||||
}
|
||||
```
|
||||
|
||||
### ChatMessage (External Interface)
|
||||
|
||||
```typescript
|
||||
interface ChatMessage {
|
||||
id?: string;
|
||||
message: string; // Display text
|
||||
originalMessage?: string; // Processed text
|
||||
sender: string;
|
||||
timestamp: FormattedDateTime | null;
|
||||
isVisible: boolean;
|
||||
context?: MessageContext;
|
||||
isErrorMessage?: boolean;
|
||||
sources?: { title: string; score: number }[];
|
||||
content?: any[];
|
||||
}
|
||||
```
|
||||
|
||||
### MessageContext
|
||||
|
||||
```typescript
|
||||
interface MessageContext {
|
||||
notes: TFile[];
|
||||
urls: string[];
|
||||
selectedTextContexts: SelectedTextContext[];
|
||||
}
|
||||
```
|
||||
|
||||
## Chat History Loading
|
||||
|
||||
### Pending Message Mechanism
|
||||
|
||||
The new architecture uses a "pending message" pattern for loading chat history:
|
||||
|
||||
```
|
||||
main.ts.loadChatHistory()
|
||||
↓
|
||||
Parse messages from file
|
||||
↓
|
||||
CopilotView.setPendingMessages()
|
||||
↓
|
||||
Chat component receives pendingMessages prop
|
||||
↓
|
||||
useEffect detects pendingMessages
|
||||
↓
|
||||
ChatUIState.loadMessages()
|
||||
↓
|
||||
onPendingMessagesProcessed() callback clears pending
|
||||
```
|
||||
|
||||
### Project-Aware Loading
|
||||
|
||||
When loading chat history:
|
||||
|
||||
1. ChatPersistenceManager filters files based on current project
|
||||
2. Only shows chat files prefixed with current project ID
|
||||
3. Non-project chats visible when no project is active
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### Unit Tests
|
||||
|
||||
- **MessageRepository**: 23 comprehensive tests including bug prevention
|
||||
- **ChatManager**: 25+ tests covering all critical functionality
|
||||
- **Component Tests**: MessageContext duplicate key prevention
|
||||
|
||||
### Bug Prevention Tests
|
||||
|
||||
1. **Context Badge Bug**: Ensures context displays correctly
|
||||
2. **Memory Synchronization**: Prevents chat memory count mismatches
|
||||
3. **Edit Message Bug**: Verifies proper context reprocessing
|
||||
4. **Duplicate Notes**: Prevents React key conflicts in context display
|
||||
|
||||
## Migration from SharedState
|
||||
|
||||
### Before (Legacy)
|
||||
|
||||
```typescript
|
||||
// Multiple sources of truth
|
||||
const sharedState = {
|
||||
currentChatMessages: ChatMessage[],
|
||||
chatHistory: ChatMessage[],
|
||||
// Complex sync logic between arrays
|
||||
}
|
||||
```
|
||||
|
||||
### After (Clean Architecture)
|
||||
|
||||
```typescript
|
||||
// Single source of truth
|
||||
const messageRepository = new MessageRepository();
|
||||
const chatManager = new ChatManager(messageRepository, ...);
|
||||
const chatUIState = new ChatUIState(chatManager);
|
||||
|
||||
// Computed views
|
||||
const displayMessages = chatUIState.getMessages(); // For UI
|
||||
const llmMessages = chatManager.getLLMMessages(); // For AI
|
||||
```
|
||||
|
||||
## Performance Considerations
|
||||
|
||||
### Memory Efficiency
|
||||
|
||||
- Single storage eliminates duplicate message objects
|
||||
- Computed views are generated on-demand
|
||||
- Context processing only when needed
|
||||
|
||||
### React Optimization
|
||||
|
||||
- Subscription-based updates minimize re-renders
|
||||
- Unique keys prevent React reconciliation issues
|
||||
- State changes are batched through ChatUIState
|
||||
|
||||
## Key Architectural Features
|
||||
|
||||
### Project Isolation Benefits
|
||||
|
||||
1. **Complete Separation**: Each project has entirely separate chat history
|
||||
2. **Automatic Management**: No user configuration needed
|
||||
3. **Seamless Switching**: Instant context switch when changing projects
|
||||
4. **Memory Efficient**: Only active project's messages in memory
|
||||
5. **Fresh Start**: Each project starts with empty chat history
|
||||
|
||||
### Persistence Integration
|
||||
|
||||
1. **Project-Aware Naming**: Files prefixed with project ID
|
||||
2. **Filtered File Lists**: Only shows relevant chat files
|
||||
3. **Consistent Format**: Same markdown format across all projects
|
||||
4. **Error Handling**: Graceful fallbacks for save/load failures
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Common Issues
|
||||
|
||||
1. **Context not updating**: Check if `updateChainMemory()` is called after edits
|
||||
2. **UI not refreshing**: Ensure `notifyListeners()` is called after state changes
|
||||
3. **Memory count mismatch**: Verify `truncateAfterMessageId()` updates chain memory
|
||||
4. **Duplicate context badges**: Check React keys in MessageContext component
|
||||
5. **Wrong project messages**: Check `getCurrentProjectId()` returns expected value
|
||||
6. **Missing chat history**: Verify project ID in filename matches current project
|
||||
|
||||
### Debug Methods
|
||||
|
||||
```typescript
|
||||
// Check message repository state
|
||||
messageRepo.getDebugInfo();
|
||||
|
||||
// Check chat manager state
|
||||
chatManager.getDebugInfo();
|
||||
|
||||
// Check LLM vs display message counts
|
||||
console.log({
|
||||
display: chatUIState.getMessages().length,
|
||||
llm: chatManager.getLLMMessages().length,
|
||||
});
|
||||
|
||||
// Check current project and repository
|
||||
const debugInfo = chatManager.getDebugInfo();
|
||||
console.log({
|
||||
currentProject: debugInfo.currentProjectId,
|
||||
totalProjects: debugInfo.projectCount,
|
||||
messagesByProject: debugInfo.messageCountByProject,
|
||||
});
|
||||
```
|
||||
|
||||
## Related Files
|
||||
|
||||
### Documentation
|
||||
|
||||
- `designdocs/CONTEXT_ENGINEERING.md` - Layered context system design and L2 auto-promotion details
|
||||
- `designdocs/MESSAGE_ARCHITECTURE.md` - This document (message management architecture)
|
||||
|
||||
### Core Implementation
|
||||
|
||||
- `src/core/MessageRepository.ts` - Message storage
|
||||
- `src/core/ChatManager.ts` - Business logic with project isolation
|
||||
- `src/state/ChatUIState.ts` - UI state management
|
||||
- `src/core/ContextManager.ts` - Context processing
|
||||
- `src/core/ChatPersistenceManager.ts` - Chat history persistence
|
||||
|
||||
### React Integration
|
||||
|
||||
- `src/components/Chat.tsx` - Main chat component
|
||||
- `src/hooks/useChatManager.ts` - React hook for ChatUIState
|
||||
- `src/components/chat-components/ChatSingleMessage.tsx` - Message display
|
||||
|
||||
### Testing
|
||||
|
||||
- `src/core/MessageRepository.test.ts` - Repository tests
|
||||
- `src/core/ChatManager.test.ts` - Manager tests
|
||||
- `src/core/MessageLifecycle.test.ts` - Complete lifecycle examples with context notes
|
||||
- `src/core/MessageLifecycle.xmltags.test.ts` - XML tag formatting tests and examples
|
||||
- `src/components/chat-components/MessageContext.test.tsx` - Context display tests
|
||||
783
designdocs/OBSIDIAN_CLI_INTEGRATION.md
Normal file
|
|
@ -0,0 +1,783 @@
|
|||
# Obsidian CLI Integration Design (MVP)
|
||||
|
||||
**Date:** 2026-02-11
|
||||
**Status:** Draft — Experimental, Desktop-Only
|
||||
**Scope:** Copilot plugin tooling (`AutonomousAgent` + `Copilot Plus` tool execution path)
|
||||
|
||||
## 1. Problem Statement
|
||||
|
||||
Obsidian now ships an official CLI (early access). Copilot already has a mature tool system, but it currently duplicates vault operations through plugin APIs and internal tool implementations. We need a low-risk path to leverage CLI capabilities without rewriting the agent loop.
|
||||
|
||||
## 2. Goals
|
||||
|
||||
1. Add Obsidian CLI capabilities with minimal changes to existing architecture.
|
||||
2. Reuse current `ToolRegistry` + LangChain native tool calling flow.
|
||||
3. Keep the first release desktop-only, safe-by-default, and observable.
|
||||
4. Ship capabilities in explicit versioned tiers (`v0` -> `v1` -> `v2`) to keep routing and validation manageable.
|
||||
|
||||
## 3. Non-Goals (MVP)
|
||||
|
||||
1. No full one-to-one wrapper for every CLI command in the first iteration.
|
||||
2. No mobile support (CLI is desktop-oriented).
|
||||
3. No prompt/system-prompt rewrites beyond normal tool metadata guidance.
|
||||
4. No dependency on interactive TUI mode in agent flows.
|
||||
|
||||
## 3a. Platform Policy
|
||||
|
||||
These tools are **desktop-only** and **experimental**.
|
||||
|
||||
- On mobile platforms, CLI tools are **not registered** in the `ToolRegistry` and are completely invisible to the user — they do not appear in tool settings, tool lists, agent reasoning, or any UI surface.
|
||||
- Registration is gated by `Platform.isDesktopApp` in `initializeBuiltinTools()`. The runtime guard in `ObsidianCliClient.runObsidianCliCommand()` provides defense-in-depth but is not the primary gating mechanism.
|
||||
- The Obsidian CLI requires `child_process.execFile`, which is only available in the desktop Electron renderer.
|
||||
|
||||
## 3b. Design Rationale — CLI Shell-Out vs Internal API
|
||||
|
||||
The CLI shell-out approach was chosen because:
|
||||
|
||||
1. **Breadth without cost**: The CLI exposes a large and growing command surface. Reimplementing each command via internal Obsidian APIs (`app.vault`, `app.metadataCache`, etc.) would require significant per-command development and maintenance effort.
|
||||
2. **Forward compatibility**: New CLI commands become available to the tool system without code changes — only the allowlist needs updating.
|
||||
3. **Safety**: `execFile` (not shell) with strict argument serialization prevents injection. A command allowlist and mutation gating limit blast radius.
|
||||
4. **Incremental adoption**: The versioned tier system (v0 → v1 → v2) allows cautious rollout, starting with read-only commands.
|
||||
|
||||
Internal API tools remain the right choice for operations that need deep integration (e.g., frontmatter processing via `app.fileManager.processFrontMatter()`). The CLI approach is complementary, not a replacement.
|
||||
|
||||
## 4. Current Architecture Fit
|
||||
|
||||
Relevant integration points:
|
||||
|
||||
- `src/tools/ToolRegistry.ts` for tool registration and metadata.
|
||||
- `src/tools/builtinTools.ts` for built-in tool definitions and initialization.
|
||||
- `src/LLMProviders/chainRunner/utils/toolExecution.ts` for execution control and user-facing tool status behavior.
|
||||
- `src/settings/model.ts` and `src/constants.ts` for defaults and persisted settings.
|
||||
- `src/settings/v2/components/ToolSettingsSection.tsx` for user tool toggles.
|
||||
|
||||
This means we can ship CLI support as one additional built-in tool (or a small tool set) without changing chat/message architecture.
|
||||
|
||||
## 5. Tool Organization: Category-Based Grouping
|
||||
|
||||
Instead of one tool per CLI command (~100 commands = too many tools) or one generic umbrella tool (too vague for LLM routing), commands are grouped into **category-based tools**. Each tool accepts a `command` parameter scoped to its category.
|
||||
|
||||
### Design Rationale
|
||||
|
||||
- **Clear semantic signals for the LLM**: User asks about daily notes → `obsidianDailyNote` tool. No ambiguity in tool selection.
|
||||
- **Scales well**: New CLI commands slot into existing category tools without adding new tool registrations.
|
||||
- **Manageable tool count**: ~10 tools total vs ~25+ individual tools or 1 opaque gateway.
|
||||
- **Per-category allowlists**: Each tool validates its `command` parameter against a scoped allowlist, limiting blast radius.
|
||||
|
||||
### v0 (Current — 2 commands, 2 tools)
|
||||
|
||||
| Tool | Commands | Notes |
|
||||
| -------------------- | ------------- | -------------------------------------------- |
|
||||
| `obsidianDailyRead` | `daily:read` | Read-only. Dedicated tool for v0 simplicity. |
|
||||
| `obsidianRandomRead` | `random:read` | Read-only. Dedicated tool for v0 simplicity. |
|
||||
|
||||
### v1 (Current — 13 commands across 7 tools)
|
||||
|
||||
All v1 tools are **read-only or direct-execution** (no confirmation UX required).
|
||||
|
||||
| Tool | Commands | Notes |
|
||||
| ---------------------- | ----------------------------------------------------------- | ----------------------------------------------------------------------------------------------- |
|
||||
| **obsidianDailyNote** | `daily:read`, `daily:append`, `daily:prepend`, `daily:path` | Append/prepend execute directly (see Write Operations Policy). Subsumes v0 `obsidianDailyRead`. |
|
||||
| **obsidianProperties** | `properties`, `property:read` | Read-only. Write commands (`property:set`, `property:remove`) deferred to v2. |
|
||||
| **obsidianTasks** | `tasks` | Read-only (task listing). Write command (`task` toggle/status) deferred to v2. |
|
||||
| **obsidianRandomRead** | `random:read` | Read-only. Standalone tool (single command). Continues from v0. |
|
||||
| **obsidianLinks** | `backlinks`, `links`, `orphans`, `unresolved` | All read-only |
|
||||
| **obsidianTemplates** | `templates`, `template:read` | Read-only. `template:insert` deferred (requires active file context). Moved from v2. |
|
||||
| **obsidianBases** | `bases`, `base:views`, `base:query`, `base:create` | Read + create. `base:create` executes directly (see Write Operations Policy). |
|
||||
|
||||
### v2 (Future — ~9 commands: 3 mutations on existing tools + 2 new tools)
|
||||
|
||||
v2 introduces **confirmation-required mutations** on existing v1 tools and adds new tool categories.
|
||||
|
||||
| Tool | Commands | Notes |
|
||||
| --------------------------------------- | --------------------------------- | ----------------------------------------------------------------------- |
|
||||
| **obsidianProperties** _(v1 extension)_ | `property:set`, `property:remove` | Light confirmation in chat before executing. Extends v1 read-only tool. |
|
||||
| **obsidianTasks** _(v1 extension)_ | `task` (toggle/done/todo/status) | Light confirmation in chat before executing. Extends v1 read-only tool. |
|
||||
| **obsidianBookmarks** | `bookmarks`, `bookmark` | `bookmark` (add) gated by mutation setting |
|
||||
|
||||
### Excluded from Tool System
|
||||
|
||||
The following CLI commands are **not exposed** to the AI agent:
|
||||
|
||||
| Category | Commands | Rationale |
|
||||
| ----------------------- | --------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ |
|
||||
| Destructive file ops | `delete`, `move`, `rename`, `create --overwrite` | Too dangerous for autonomous agent use |
|
||||
| Plugin/theme management | `plugin:*`, `theme:*`, `snippet:*` | Not an AI task, security risk |
|
||||
| Sync & history | `sync:*`, `history:restore`, `diff` | User-managed operations, data loss risk |
|
||||
| UI/workspace control | `tabs`, `tab:open`, `workspace`, `open`, `daily` (open variant) | UI-only, no data value for LLM |
|
||||
| System commands | `reload`, `restart`, `version`, `vault`, `vaults` | Not useful for agent workflows |
|
||||
| Developer tools | `eval`, `dev:*`, `devtools` | Arbitrary code execution risk |
|
||||
| Niche metadata | `aliases`, `wordcount`, `recents`, `hotkeys`, `commands` | Low AI synergy |
|
||||
| Search | `search`, `search:context`, `search:open` | Redundant with Copilot's existing keyword + semantic search (`localSearch`) |
|
||||
| File read | `read` | Redundant with Copilot's existing `readNote` tool (see Tool Disambiguation) |
|
||||
| Tag listing | `tags` | Redundant with Copilot's existing `getTagList` tool (see Tool Disambiguation) |
|
||||
| Arbitrary file writes | `append`, `prepend`, `create` | File modifications beyond daily notes should go through the existing Composer tool (`writeToFile`/`replaceInFile`) |
|
||||
|
||||
### Write Operations Policy
|
||||
|
||||
| Operation | Execution model | Tier | Rationale |
|
||||
| --------------------------------- | ---------------------------------------------- | ---- | ----------------------------------------------------------------------------------------------------- |
|
||||
| **Daily note append/prepend** | Direct execution, show result in chat response | v1 | User explicitly asked for the action; daily notes are append-only by nature and low-risk |
|
||||
| **Base create** | Direct execution, show result in chat response | v1 | User explicitly asked to add an item; creates a new note matching Base filters, additive and low-risk |
|
||||
| **Arbitrary file append/prepend** | Excluded — use Composer tool | — | File modifications beyond daily notes need the Composer diff/preview UX for safety |
|
||||
| **Property set/remove** | Light confirmation in chat before executing | v2 | Metadata changes are reversible but should be intentional; deferred to validate read-only tools first |
|
||||
| **Task toggle/status** | Light confirmation in chat before executing | v2 | Status changes are reversible but should be intentional; deferred to validate read-only tools first |
|
||||
|
||||
### Tool Disambiguation
|
||||
|
||||
CLI tools are **complementary** to existing internal tools, not replacements. Several CLI commands were evaluated and intentionally excluded because Copilot already has superior internal implementations:
|
||||
|
||||
| CLI Command | Existing Internal Tool | Why Internal Wins |
|
||||
| ----------------------------- | ------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `read` | `readNote` | In-process (`app.vault.cachedRead`), 200-line chunking, multi-strategy path resolution (wikilink, basename, partial match), linked notes extraction, mtime metadata. CLI `read` spawns a process, returns raw text, and requires exact paths. |
|
||||
| `tags` | `getTagList` | In-process (`app.metadataCache`), structured JSON with occurrence counts, frontmatter/inline breakdown, progressive size limiting (500KB cap), configurable `maxEntries`. CLI `tags` returns unstructured text with no filtering. |
|
||||
| `search`, `search:context` | `localSearch` | Hybrid keyword + semantic search with BM25, query expansion, reranking, time range filtering, tag-aware retrieval. CLI search is basic text matching. |
|
||||
| `append`, `prepend`, `create` | `writeToFile` / `replaceInFile` | Composer diff/preview UX for safety, line-ending normalization, SEARCH/REPLACE blocks, auto-accept setting. Exception: `daily:append`/`daily:prepend` use CLI directly (low-risk, append-only). |
|
||||
|
||||
**Prompt instruction guidelines for CLI tools:**
|
||||
|
||||
- Each CLI tool's `customPromptInstructions` must include explicit disambiguation guidance directing the LLM to the correct tool.
|
||||
- Example: "Use `readNote` for reading specific notes by path. Use `obsidianDailyNote` for daily note operations (read, append, prepend). Use `obsidianRandomRead` for picking a random note."
|
||||
- When an existing internal tool and a CLI tool could both handle a request, the internal tool should be preferred unless the CLI tool provides unique capability (e.g., daily note path resolution, random note selection, backlink traversal).
|
||||
|
||||
## 6. Implementation Design
|
||||
|
||||
### 6.1 Service Layer: `ObsidianCliClient`
|
||||
|
||||
Located at `src/services/obsidianCli/ObsidianCliClient.ts`. Responsible for:
|
||||
|
||||
1. CLI availability/version checks.
|
||||
2. Safe command execution via `execFile` (not shell).
|
||||
3. Argument serialization to `parameter=value` and boolean flags.
|
||||
4. Timeout + output-size limits.
|
||||
5. Structured error mapping for tool responses.
|
||||
6. Fallback binary resolution (tries `obsidian` → known macOS app paths).
|
||||
|
||||
Key guardrails:
|
||||
|
||||
- No shell interpolation.
|
||||
- Per-tool command allowlist.
|
||||
- Desktop-only runtime guard.
|
||||
|
||||
### 6.2 Tool Layer
|
||||
|
||||
Each category tool is a LangChain `StructuredTool` with a zod schema. Tools are registered conditionally in `src/tools/builtinTools.ts` via `registerCliTools()`, gated by `Platform.isDesktopApp`.
|
||||
|
||||
### 6.3 Settings
|
||||
|
||||
Planned settings fields:
|
||||
|
||||
1. `obsidianCliAllowMutations: boolean` (default `false`) — gates write commands across all CLI tools.
|
||||
2. `obsidianCliTimeoutMs: number` (default `15000`) — per-command timeout.
|
||||
3. `obsidianCliPath: string` (default `"obsidian"`) — custom binary path override.
|
||||
|
||||
## 7. Execution + UX Rules
|
||||
|
||||
1. If CLI unavailable, return a clear actionable tool error.
|
||||
2. If running on mobile, return unsupported-platform error.
|
||||
3. Surface tool result summaries in existing tool banners/reasoning stream.
|
||||
4. Keep file-modifying behavior conservative until explicit mutation rollout.
|
||||
|
||||
## 8. Testing Plan
|
||||
|
||||
Unit tests:
|
||||
|
||||
1. Argument serialization (`parameter=value`, booleans, multiline escaping handling).
|
||||
2. Allowlist enforcement and mutation gating.
|
||||
3. Timeout/error mapping behavior.
|
||||
4. Desktop/mobile guards.
|
||||
|
||||
Integration tests (mocked process):
|
||||
|
||||
1. Successful command execution output.
|
||||
2. Non-zero exit behavior.
|
||||
3. Large output truncation/handling.
|
||||
|
||||
Manual validation:
|
||||
|
||||
1. Run read-only commands from agent mode and verify response quality.
|
||||
2. Verify settings toggles affect tool availability correctly.
|
||||
|
||||
## 9. Rollout Plan
|
||||
|
||||
### Phase 0: Design + scaffolding (done)
|
||||
|
||||
- Design doc, `ObsidianCliClient` service, `obsidianDailyRead`/`obsidianRandomRead` tools.
|
||||
- Desktop-only gating via `Platform.isDesktopApp` in `registerCliTools()`.
|
||||
- Tests for arg serialization, fallback binary resolution, tool wrappers.
|
||||
|
||||
### Phase 1: v0 release (current)
|
||||
|
||||
- Ship `daily:read` and `random:read` as dedicated read-only tools.
|
||||
- Validate CLI reliability and UX across desktop platforms.
|
||||
|
||||
### Phase 2: v1 expansion
|
||||
|
||||
- Refactor v0 `obsidianDailyRead` into category-based `obsidianDailyNote` (read, append, prepend, path).
|
||||
- Keep `obsidianRandomRead` as standalone tool (single command).
|
||||
- Add `obsidianProperties` (read-only), `obsidianTasks` (read-only), `obsidianLinks` tools.
|
||||
- All v1 tools are read-only or direct-execution — no confirmation UX needed.
|
||||
- Add per-tool command allowlists and prompt disambiguation guidance.
|
||||
|
||||
### Phase 3: v2 expansion
|
||||
|
||||
- Add confirmation-required mutations to existing v1 tools: `property:set`, `property:remove`, `task` toggle/status.
|
||||
- Implement mutation gating via `obsidianCliAllowMutations` setting and light confirmation UX.
|
||||
- Add new tool categories: `obsidianTemplates`, `obsidianBases`, `obsidianBookmarks`.
|
||||
|
||||
## 10. Risks and Mitigations
|
||||
|
||||
1. CLI behavior/version drift.
|
||||
Mitigation: version checks + graceful fallback.
|
||||
2. Security risks from command injection.
|
||||
Mitigation: `execFile`, allowlist, strict param serializer.
|
||||
3. UX confusion between existing tools and CLI-backed actions.
|
||||
Mitigation: clear tool naming + incremental command scope.
|
||||
|
||||
## 11. Open Questions
|
||||
|
||||
### Resolved
|
||||
|
||||
1. ~~Should `obsidianCli` be exposed in standard tool settings for all users, or hidden behind a feature flag first?~~
|
||||
**Resolved**: CLI tools are registered in `ToolRegistry` like any other tool, gated by `Platform.isDesktopApp`. No separate feature flag — they appear in tool settings on desktop, invisible on mobile.
|
||||
|
||||
2. ~~Do we want a dedicated category for CLI-backed tools?~~
|
||||
**Resolved**: No dedicated category. CLI tools use category `"file"` alongside existing file tools. They are distinguished by their `id` prefix (`obsidian*`) and `displayName` suffix `(CLI)`.
|
||||
|
||||
3. ~~Should we prefer existing internal tools over CLI for certain operations (for consistency/performance)?~~
|
||||
**Resolved**: Yes. Internal tools are preferred when they exist. See Tool Disambiguation section — `readNote` over CLI `read`, `getTagList` over CLI `tags`, `localSearch` over CLI `search`, Composer over CLI file writes. CLI tools are only used for capabilities without an internal equivalent.
|
||||
|
||||
### Open
|
||||
|
||||
1. What minimum CLI version should be required for initial support?
|
||||
2. How should the agent reliably choose between similar tools when both could handle a request? For example, `daily:append`/`daily:prepend` vs the Composer tool (`writeToFile`/`replaceInFile`) when the user says "add something to my daily note." Current approach is prompt-instruction disambiguation, but this depends on LLM adherence to instructions. Alternatives: remove overlapping commands entirely, or add runtime routing that intercepts and redirects.
|
||||
3. Can we reliably resolve the Obsidian CLI binary path across platforms? Current approach: try `obsidian` on PATH → env vars (`OBSIDIAN_CLI_BINARY`, `OBSIDIAN_CLI_PATH`) → macOS fallback paths (`/Applications/Obsidian.app/Contents/MacOS/obsidian`). Windows and Linux fallback paths are not yet implemented. If the CLI is not on PATH and no env var is set, the tool fails. Should we add a settings field for manual path override, or auto-detect from known install locations per platform?
|
||||
|
||||
---
|
||||
|
||||
## Appendix A: V1 CLI Command Reference
|
||||
|
||||
All commands are invoked as `obsidian <command> [params...]`. Output is **plain text** (no `format=json`) — LLMs consume text natively without the token overhead of JSON structure.
|
||||
|
||||
Global parameter available on all commands: `vault=<name>` (targets a specific vault; omit for default).
|
||||
|
||||
### A.1 `obsidianDailyNote` — Daily Note Operations
|
||||
|
||||
#### `daily:read`
|
||||
|
||||
Read today's daily note content.
|
||||
|
||||
```
|
||||
obsidian daily:read
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| --------- | -------- | ---------------------------------------------- |
|
||||
| _(none)_ | | No parameters. Reads the daily note for today. |
|
||||
|
||||
**Output**: Full markdown content of today's daily note. Empty string if no daily note exists.
|
||||
|
||||
```
|
||||
# 2026-03-03
|
||||
|
||||
## Tasks
|
||||
- [ ] Review PR #2181
|
||||
- [x] Update design doc
|
||||
|
||||
## Notes
|
||||
Meeting with Alice about CLI integration...
|
||||
```
|
||||
|
||||
#### `daily:path`
|
||||
|
||||
Get the vault-relative file path of today's daily note.
|
||||
|
||||
```
|
||||
obsidian daily:path
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| --------- | -------- | -------------- |
|
||||
| _(none)_ | | No parameters. |
|
||||
|
||||
**Output**: Single line — the vault-relative path.
|
||||
|
||||
```
|
||||
2026-03-03.md
|
||||
```
|
||||
|
||||
This is the only way to discover the daily note path without knowing the user's daily note folder/date format configuration.
|
||||
|
||||
#### `daily:append`
|
||||
|
||||
Append content to the end of today's daily note. Creates the daily note if it doesn't exist.
|
||||
|
||||
```
|
||||
obsidian daily:append content="- Meeting with Alice at 3pm"
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------------- | -------- | ----------------------------------------------- |
|
||||
| `content=<text>` | Yes | Text to append. |
|
||||
| `inline` | No | Boolean flag. Append without a leading newline. |
|
||||
|
||||
**Output**: Empty on success. The content is added at the end of the file.
|
||||
|
||||
**Note**: `open` and `paneType` parameters are accepted by the CLI but are not passed by the tool (UI-only, no value for agent).
|
||||
|
||||
#### `daily:prepend`
|
||||
|
||||
Prepend content to the beginning of today's daily note (after frontmatter). Creates the daily note if it doesn't exist.
|
||||
|
||||
```
|
||||
obsidian daily:prepend content="## Morning Standup"
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------------- | -------- | ------------------------------------------------- |
|
||||
| `content=<text>` | Yes | Text to prepend. |
|
||||
| `inline` | No | Boolean flag. Prepend without a trailing newline. |
|
||||
|
||||
**Output**: Empty on success.
|
||||
|
||||
---
|
||||
|
||||
### A.2 `obsidianProperties` — Note Property Access
|
||||
|
||||
#### `properties`
|
||||
|
||||
List frontmatter properties. Can operate vault-wide or on a specific note.
|
||||
|
||||
**Vault-wide** (list all property names used across the vault):
|
||||
|
||||
```
|
||||
obsidian properties
|
||||
```
|
||||
|
||||
```
|
||||
aliases
|
||||
author
|
||||
cssclasses
|
||||
date
|
||||
tags
|
||||
title
|
||||
```
|
||||
|
||||
**For a specific note** (list that note's property key-value pairs):
|
||||
|
||||
```
|
||||
obsidian properties file="Rewrite as tweet"
|
||||
```
|
||||
|
||||
```
|
||||
copilot-command-context-menu-enabled: false
|
||||
copilot-command-slash-enabled: false
|
||||
copilot-command-context-menu-order: 90
|
||||
copilot-command-model-key: ""
|
||||
copilot-command-last-used: 0
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ------------- | -------- | --------------------------------------------------------- |
|
||||
| `file=<name>` | No | Target file by name (without extension). |
|
||||
| `path=<path>` | No | Target file by vault-relative path. |
|
||||
| `name=<name>` | No | Get count for a specific property name (vault-wide mode). |
|
||||
| `counts` | No | Include occurrence counts (vault-wide mode). |
|
||||
| `sort=count` | No | Sort by count instead of name (vault-wide mode). |
|
||||
| `total` | No | Return only the property count. |
|
||||
|
||||
**Output (vault-wide)**: One property name per line, alphabetically sorted by default. With `counts`, format is `name: count`. With `total`, a single number.
|
||||
|
||||
**Output (per-file)**: `key: value` pairs, one per line (YAML-like).
|
||||
|
||||
#### `property:read`
|
||||
|
||||
Read a single property value from a specific note.
|
||||
|
||||
```
|
||||
obsidian property:read name="tags" file="My Note"
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ------------- | -------- | ----------------------------------- |
|
||||
| `name=<name>` | Yes | Property name to read. |
|
||||
| `file=<name>` | No | Target file by name. |
|
||||
| `path=<path>` | No | Target file by vault-relative path. |
|
||||
|
||||
**Output**: The raw property value. For arrays, comma-separated. For strings, the plain value.
|
||||
|
||||
```
|
||||
90
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### A.3 `obsidianTasks` — Task Listing
|
||||
|
||||
#### `tasks`
|
||||
|
||||
List tasks across the vault with filtering options.
|
||||
|
||||
```
|
||||
obsidian tasks todo
|
||||
obsidian tasks file="Project Plan" verbose
|
||||
obsidian tasks daily
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ----------------- | -------- | ---------------------------------------------------------------- |
|
||||
| `file=<name>` | No | Filter by file name. |
|
||||
| `path=<path>` | No | Filter by file path. |
|
||||
| `todo` | No | Show only incomplete tasks. |
|
||||
| `done` | No | Show only completed tasks. |
|
||||
| `status="<char>"` | No | Filter by status character (e.g., `status="/"` for in-progress). |
|
||||
| `daily` | No | Show tasks from today's daily note. |
|
||||
| `verbose` | No | Group tasks by file with line numbers. |
|
||||
| `total` | No | Return only the task count. |
|
||||
|
||||
**Output (default text)**: One task per line, markdown checkbox format.
|
||||
|
||||
```
|
||||
- [ ] Review PR #2181
|
||||
- [ ] Update design doc
|
||||
- [x] Write CLI client tests
|
||||
```
|
||||
|
||||
**Output (verbose)**: Tasks grouped under file headings with line numbers.
|
||||
|
||||
```
|
||||
Projects/launch-plan.md
|
||||
L12: - [ ] Review PR #2181
|
||||
L15: - [x] Write CLI client tests
|
||||
|
||||
Daily/2026-03-03.md
|
||||
L8: - [ ] Update design doc
|
||||
```
|
||||
|
||||
**Output (total)**: Single number.
|
||||
|
||||
```
|
||||
3
|
||||
```
|
||||
|
||||
**Empty result**: `No tasks found.`
|
||||
|
||||
---
|
||||
|
||||
### A.4 `obsidianRandomRead` — Random Note
|
||||
|
||||
#### `random:read`
|
||||
|
||||
Read a randomly selected markdown note from the vault.
|
||||
|
||||
```
|
||||
obsidian random:read
|
||||
obsidian random:read folder="Ideas"
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| --------------- | -------- | ------------------------------------- |
|
||||
| `folder=<path>` | No | Limit selection to a specific folder. |
|
||||
|
||||
**Output**: Full markdown content of the randomly selected note. A different note is returned each invocation.
|
||||
|
||||
**Empty result**: `No markdown files found.` (when folder is empty or doesn't exist).
|
||||
|
||||
---
|
||||
|
||||
### A.5 `obsidianLinks` — Link Graph Queries
|
||||
|
||||
#### `backlinks`
|
||||
|
||||
List notes that link TO a given file (incoming links).
|
||||
|
||||
```
|
||||
obsidian backlinks file="My Note"
|
||||
obsidian backlinks path="Projects/plan.md" counts
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ------------- | -------- | ------------------------------------ |
|
||||
| `file=<name>` | No | Target file by name. |
|
||||
| `path=<path>` | No | Target file by vault-relative path. |
|
||||
| `counts` | No | Include link counts per source file. |
|
||||
| `total` | No | Return only the backlink count. |
|
||||
|
||||
**Output (default TSV)**: One source file per line.
|
||||
|
||||
```
|
||||
Projects/roadmap.md
|
||||
Daily/2026-03-01.md
|
||||
```
|
||||
|
||||
**Output (counts)**: Source file with link count.
|
||||
|
||||
```
|
||||
Projects/roadmap.md 3
|
||||
Daily/2026-03-01.md 1
|
||||
```
|
||||
|
||||
**Output (total)**: Single number.
|
||||
|
||||
**Empty result**: `No backlinks found.`
|
||||
|
||||
#### `links`
|
||||
|
||||
List outgoing links FROM a given file.
|
||||
|
||||
```
|
||||
obsidian links file="My Note"
|
||||
obsidian links path="Projects/plan.md" total
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ------------- | -------- | ----------------------------------- |
|
||||
| `file=<name>` | No | Source file by name. |
|
||||
| `path=<path>` | No | Source file by vault-relative path. |
|
||||
| `total` | No | Return only the link count. |
|
||||
|
||||
**Output**: One link target per line.
|
||||
|
||||
```
|
||||
Projects/roadmap.md
|
||||
Ideas/brainstorm.md
|
||||
```
|
||||
|
||||
**Empty result**: `No links found.`
|
||||
|
||||
#### `orphans`
|
||||
|
||||
List files with no incoming links (not linked from any other note).
|
||||
|
||||
```
|
||||
obsidian orphans
|
||||
obsidian orphans total
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| --------- | -------- | ------------------------------------------------ |
|
||||
| `total` | No | Return only the orphan count. |
|
||||
| `all` | No | Include non-markdown files (images, PDFs, etc.). |
|
||||
|
||||
**Output**: One file path per line.
|
||||
|
||||
```
|
||||
2026-03-03.md
|
||||
BOT/DailyAIDigest/2026-02-26-Daily-AI-Digest.md
|
||||
copilot/copilot-conversations/hello@20260302_145233.md
|
||||
DemoCanvas.canvas
|
||||
```
|
||||
|
||||
**Output (total)**: Single number (e.g., `84`).
|
||||
|
||||
#### `unresolved`
|
||||
|
||||
List wikilinks that don't resolve to any existing file in the vault.
|
||||
|
||||
```
|
||||
obsidian unresolved
|
||||
obsidian unresolved counts verbose
|
||||
obsidian unresolved total
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| --------- | -------- | ---------------------------------------------------- |
|
||||
| `counts` | No | Include how many times each unresolved link appears. |
|
||||
| `verbose` | No | Include source file for each unresolved link. |
|
||||
| `total` | No | Return only the unresolved link count. |
|
||||
|
||||
**Output (default TSV)**: One unresolved link target per line.
|
||||
|
||||
```
|
||||
Nonexistent Note
|
||||
Old Project Reference
|
||||
meeting-notes-2025
|
||||
```
|
||||
|
||||
**Output (counts)**: Link target with occurrence count.
|
||||
|
||||
```
|
||||
Nonexistent Note 5
|
||||
Old Project Reference 2
|
||||
```
|
||||
|
||||
**Output (verbose)**: Link target with source files.
|
||||
|
||||
```
|
||||
Nonexistent Note Projects/roadmap.md
|
||||
Nonexistent Note Daily/2026-03-01.md
|
||||
Old Project Reference Archive/cleanup.md
|
||||
```
|
||||
|
||||
**Output (total)**: Single number (e.g., `771`).
|
||||
|
||||
---
|
||||
|
||||
### A.6 `obsidianTemplates` — Template Listing and Reading
|
||||
|
||||
#### `templates`
|
||||
|
||||
List all available template names in the configured templates folder.
|
||||
|
||||
```
|
||||
obsidian templates
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| --------- | -------- | ---------------------------------------- |
|
||||
| _(none)_ | | No parameters. Lists all template names. |
|
||||
|
||||
**Output**: One template name per line.
|
||||
|
||||
```
|
||||
Daily Note
|
||||
Meeting Notes
|
||||
Project Plan
|
||||
Weekly Review
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
#### `template:read`
|
||||
|
||||
Read a template's content with variable placeholders resolved.
|
||||
|
||||
```
|
||||
obsidian template:read name="Daily Note"
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ------------- | -------- | ------------------------------------------- |
|
||||
| `name=<name>` | Yes | Template name (as returned by `templates`). |
|
||||
|
||||
**Output**: Full markdown content of the template.
|
||||
|
||||
```
|
||||
# {{date}}
|
||||
|
||||
## Tasks
|
||||
- [ ]
|
||||
|
||||
## Notes
|
||||
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### A.7 `obsidianBases` — Base Database Queries
|
||||
|
||||
#### `bases`
|
||||
|
||||
List all Base (database) files in the vault.
|
||||
|
||||
```
|
||||
obsidian bases
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| --------- | -------- | ------------------------------------ |
|
||||
| `total` | No | Return only the count of Base files. |
|
||||
|
||||
**Output**: One Base file per line.
|
||||
|
||||
```
|
||||
Contacts.base
|
||||
Projects.base
|
||||
Tasks.base
|
||||
```
|
||||
|
||||
**Output (total)**: Single number.
|
||||
|
||||
#### `base:views`
|
||||
|
||||
List views defined in a Base file.
|
||||
|
||||
```
|
||||
obsidian base:views file="Projects"
|
||||
obsidian base:views path="Databases/Projects.base"
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ------------- | -------- | --------------------------------------------- |
|
||||
| `file=<name>` | No\* | Target Base file by name (without extension). |
|
||||
| `path=<path>` | No\* | Target Base file by vault-relative path. |
|
||||
|
||||
\* One of `file` or `path` is required.
|
||||
|
||||
**Output**: One view name per line.
|
||||
|
||||
```
|
||||
All Items
|
||||
By Status
|
||||
Kanban
|
||||
```
|
||||
|
||||
#### `base:create`
|
||||
|
||||
Create a new item (row) in a Base. The created item is a new markdown note that matches the Base's filter criteria.
|
||||
|
||||
```
|
||||
obsidian base:create file="Library" name="Dune Messiah" content="A book by Frank Herbert"
|
||||
obsidian base:create path="Databases/Projects.base" view="Active" name="New Feature"
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------------- | -------- | ------------------------------------------------------------- |
|
||||
| `file=<name>` | No\* | Target Base file by name (without extension). |
|
||||
| `path=<path>` | No\* | Target Base file by vault-relative path. |
|
||||
| `view=<name>` | No | View to add the item to. Omit for default view. |
|
||||
| `name=<name>` | No | File name for the created note. Omit for auto-generated name. |
|
||||
| `content=<text>` | No | Initial markdown content for the note. |
|
||||
|
||||
\* One of `file` or `path` is required.
|
||||
|
||||
**Output**: Confirmation message with the path of the created note.
|
||||
|
||||
**Note**: `open` and `newtab` parameters are accepted by the CLI but not passed by the tool (UI-only, no value for agent).
|
||||
|
||||
---
|
||||
|
||||
#### `base:query`
|
||||
|
||||
Query data from a Base view.
|
||||
|
||||
```
|
||||
obsidian base:query file="Projects" view="All Items"
|
||||
obsidian base:query path="Databases/Projects.base" format=csv
|
||||
```
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| -------------- | -------- | --------------------------------------------------- |
|
||||
| `file=<name>` | No\* | Target Base file by name (without extension). |
|
||||
| `path=<path>` | No\* | Target Base file by vault-relative path. |
|
||||
| `view=<name>` | No | View name to query. Omit for default view. |
|
||||
| `format=<fmt>` | No | Output format (e.g., `csv`). Omit for default text. |
|
||||
| `total` | No | Return only the row count. |
|
||||
|
||||
\* One of `file` or `path` is required.
|
||||
|
||||
**Output (default text)**: Tabular data, one row per line.
|
||||
|
||||
**Output (csv)**: CSV-formatted data.
|
||||
|
||||
```
|
||||
Name,Status
|
||||
Alpha,Active
|
||||
Beta,Done
|
||||
```
|
||||
|
||||
**Output (total)**: Single number.
|
||||
|
||||
---
|
||||
|
||||
### A.8 Error Responses
|
||||
|
||||
All commands return consistent error formats:
|
||||
|
||||
| Condition | Output |
|
||||
| ---------------------- | -------------------------------------------------------------------------------------------------- |
|
||||
| File not found | `Error: File "path/to/file.md" not found.` |
|
||||
| Missing required param | `Error: Missing required parameter: name=<name>` with usage line |
|
||||
| No results | Command-specific empty message (e.g., `No tasks found.`, `No backlinks found.`, `No links found.`) |
|
||||
| CLI binary not found | Process error code `ENOENT` — handled by `ObsidianCliClient` fallback resolution |
|
||||
| Timeout | Process killed after `timeoutMs` — handled by `ObsidianCliClient` |
|
||||
363
designdocs/TOOLS.md
Normal file
|
|
@ -0,0 +1,363 @@
|
|||
# Tool System Documentation
|
||||
|
||||
## Overview
|
||||
|
||||
The Copilot tool system uses a centralized registry pattern that makes it easy to add new tools, including future MCP (Model Context Protocol) tools. All tools are managed through a singleton `ToolRegistry` that provides a unified interface for tool discovery, configuration, and execution.
|
||||
|
||||
## Tool Prompt Architecture
|
||||
|
||||
### How Tool Instructions Flow to the LLM
|
||||
|
||||
The system uses **native tool calling** via LangChain's `bindTools()` for tool invocation. Tool results are formatted as context in a layered approach:
|
||||
|
||||
1. **Tool Schema Descriptions** (Zod schemas in tool implementations)
|
||||
|
||||
- Defines parameter formats, rules, and validation
|
||||
- Provided to LLM via `bindTools()` for native tool calling
|
||||
|
||||
2. **Custom Prompt Instructions** (in `builtinTools.ts`)
|
||||
|
||||
- Behavioral guidance for when and how to use tools
|
||||
- Special requirements (e.g., "always provide salientTerms")
|
||||
|
||||
3. **Model-Specific Adaptations** (in `modelAdapter.ts`)
|
||||
- Last resort for model-specific quirks
|
||||
|
||||
### Layered Prompt Integration
|
||||
|
||||
- `ContextManager` promotes user-attached artifacts from earlier turns into **L2 (Context Library)**, so every chain runner starts with the same cacheable system prefix.
|
||||
- When a tool executes during the current turn, its result payload is prepended to the **user message** (L3 + L5) using `renderCiCMessage(...)`. Nothing is injected into the system message, keeping L1/L2 stable.
|
||||
- `LayerToMessagesConverter.convert(envelope, { includeSystemMessage: true, mergeUserContent: true })` materializes the base messages; runners then append tool results before sending to the model.
|
||||
- `promptPayloadRecorder` inspects the final payload and highlights tool blocks in its layered view, making it easy to debug the L1-L5 structure.
|
||||
|
||||
### Why Two Layers: Schema vs Custom Instructions
|
||||
|
||||
**Key Difference**: Schema descriptions document parameters via Zod, while custom instructions provide behavioral guidance.
|
||||
|
||||
1. **Clear Separation**
|
||||
|
||||
- **Schema**: Parameter documentation (types, formats, rules) via Zod - used by `bindTools()`
|
||||
- **Custom Instructions**: Behavioral guidance on when and how to use the tool
|
||||
|
||||
2. **MCP Compatibility**
|
||||
|
||||
- External tools provide immutable schemas
|
||||
- We add custom instructions without modifying their code
|
||||
|
||||
3. **Example**
|
||||
|
||||
```typescript
|
||||
// Schema (provided to LLM via bindTools())
|
||||
const searchSchema = z.object({
|
||||
query: z.string().min(1).describe("The search query"),
|
||||
salientTerms: z.array(z.string()).describe("Keywords to find in notes"),
|
||||
});
|
||||
|
||||
// Custom Instructions (behavioral guidance)
|
||||
customPromptInstructions: `
|
||||
When searching notes:
|
||||
- Always provide salientTerms extracted from the user's query
|
||||
- Use getTimeRangeMs first for time-based queries
|
||||
- Examine relevance scores before using results
|
||||
`;
|
||||
```
|
||||
|
||||
### Best Practices
|
||||
|
||||
1. **Schema Descriptions: Parameter Documentation Only**
|
||||
|
||||
- Document parameter types, formats, and validation rules via Zod
|
||||
- Schemas are automatically provided to LLM via `bindTools()`
|
||||
- Focus on the data contract
|
||||
|
||||
2. **Custom Instructions: Behavioral Guidance**
|
||||
|
||||
- Explain when to use this tool vs alternatives
|
||||
- Show common usage patterns and requirements
|
||||
- Include tips for better results (e.g., "use getTimeRangeMs before localSearch")
|
||||
|
||||
3. **Model Adapters: Model-Specific Fixes**
|
||||
- Only for persistent model-specific failures
|
||||
- Keep minimal and targeted
|
||||
|
||||
### localSearch CiC Prompting Flow
|
||||
|
||||
- CiC: Corpus in Context https://arxiv.org/pdf/2406.13121
|
||||
- **Instruction First**: `CopilotPlusChainRunner` now assembles the localSearch payload via `buildLocalSearchInnerContent`, ensuring citation guidance (e.g., `<guidance>` rules) tops the XML block before any documents.
|
||||
- **Documents Next**: Search hits are serialized once through `formatSearchResultsForLLM`; the helper simply appends them after guidance, keeping the documents section untouched but clearly separated.
|
||||
- **Question Last**: `renderCiCMessage` formats the final prompt so any context precedes the user's original query; this matches the CiC recommendation for instruction → context → query ordering.
|
||||
- **CopilotPlus**: Uses `LayerToMessagesConverter` which adds `[User query]:` label when merging L3+L5 content from envelope
|
||||
- **AutonomousAgent**: Uses `ensureCiCOrderingWithQuestion` which adds `[User query]:` label to clearly separate tool results from original query in iterative loop
|
||||
- **Consistency**: Both chains use the same `[User query]:` label format for uniform prompting across the codebase
|
||||
- **Reusable Wrapping**: `wrapLocalSearchPayload` centralizes the `<localSearch>` tag creation (including optional `timeRange`), making the layout reusable for future chains without copying string glue.
|
||||
|
||||
## Current Implementation
|
||||
|
||||
### Core Files
|
||||
|
||||
- `src/tools/ToolRegistry.ts` - Central registry for all tools
|
||||
- `src/tools/builtinTools.ts` - Built-in tool definitions and initialization
|
||||
- `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts` - Tool execution in the agent
|
||||
- `src/settings/v2/components/ToolSettingsSection.tsx` - Settings UI for tool configuration
|
||||
|
||||
### Tool Registry Pattern
|
||||
|
||||
The `ToolRegistry` is a singleton that manages all tools:
|
||||
|
||||
```typescript
|
||||
class ToolRegistry {
|
||||
static getInstance(): ToolRegistry;
|
||||
register(definition: ToolDefinition): void;
|
||||
registerAll(definitions: ToolDefinition[]): void;
|
||||
getAllTools(): ToolDefinition[];
|
||||
getEnabledTools(enabledToolIds: Set<string>, vaultAvailable: boolean): SimpleTool<any, any>[];
|
||||
getToolsByCategory(): Map<string, ToolDefinition[]>;
|
||||
getConfigurableTools(): ToolDefinition[];
|
||||
getToolMetadata(id: string): ToolMetadata | undefined;
|
||||
clear(): void;
|
||||
}
|
||||
```
|
||||
|
||||
### Tool Definition Structure
|
||||
|
||||
```typescript
|
||||
interface ToolDefinition {
|
||||
tool: SimpleTool<any, any>; // The actual tool implementation
|
||||
metadata: ToolMetadata; // UI and configuration metadata
|
||||
}
|
||||
|
||||
interface ToolMetadata {
|
||||
id: string; // Unique identifier
|
||||
displayName: string; // Shown in UI
|
||||
description: string; // Help text
|
||||
category: "search" | "time" | "file" | "media" | "mcp" | "custom";
|
||||
isAlwaysEnabled?: boolean; // If true, not configurable (e.g., time tools)
|
||||
requiresVault?: boolean; // Needs vault access
|
||||
customPromptInstructions?: string; // Tool-specific prompts
|
||||
}
|
||||
```
|
||||
|
||||
## Adding a New Built-in Tool
|
||||
|
||||
### 1. Implement the Tool
|
||||
|
||||
Create your tool following the `SimpleTool` interface:
|
||||
|
||||
```typescript
|
||||
// Example: New built-in tool
|
||||
import { z } from "zod";
|
||||
import { SimpleTool } from "./SimpleTool";
|
||||
|
||||
export const myNewTool: SimpleTool<{ input: string }, { result: string }> = {
|
||||
name: "myNewTool",
|
||||
description: "Description for the LLM to understand when to use this tool",
|
||||
schema: z.object({
|
||||
input: z.string().describe("The input parameter description"),
|
||||
}),
|
||||
func: async (params) => {
|
||||
// Tool implementation
|
||||
const result = await performOperation(params.input);
|
||||
return { result };
|
||||
},
|
||||
};
|
||||
```
|
||||
|
||||
### 2. Add to Built-in Tools
|
||||
|
||||
Update `src/tools/builtinTools.ts`:
|
||||
|
||||
```typescript
|
||||
export const BUILTIN_TOOLS: ToolDefinition[] = [
|
||||
// ... existing tools ...
|
||||
{
|
||||
tool: myNewTool,
|
||||
metadata: {
|
||||
id: "myNewTool",
|
||||
displayName: "My New Tool",
|
||||
description: "User-friendly description for settings UI",
|
||||
category: "custom", // Choose appropriate category
|
||||
// Optional flags:
|
||||
isAlwaysEnabled: false, // Set true if tool should always be available
|
||||
requiresVault: true, // Set true if tool needs vault access
|
||||
customPromptInstructions: "Special instructions for the AI when using this tool",
|
||||
},
|
||||
},
|
||||
];
|
||||
```
|
||||
|
||||
### 3. Update Default Settings (if configurable)
|
||||
|
||||
If the tool is configurable (not always-enabled), add its ID to the default enabled tools in `src/constants.ts`:
|
||||
|
||||
```typescript
|
||||
autonomousAgentEnabledToolIds: [
|
||||
"localSearch",
|
||||
"webSearch",
|
||||
"pomodoro",
|
||||
"youtubeTranscription",
|
||||
"writeToFile",
|
||||
"myNewTool" // Add your tool ID here
|
||||
],
|
||||
```
|
||||
|
||||
## Adding MCP Tools (Future Implementation)
|
||||
|
||||
### 1. MCP Tool Wrapper
|
||||
|
||||
Create a wrapper to convert MCP tools to the SimpleTool interface:
|
||||
|
||||
```typescript
|
||||
function createMcpToolWrapper(serverName: string, mcpTool: McpTool): SimpleTool<any, any> {
|
||||
return {
|
||||
name: `${serverName}_${mcpTool.name}`,
|
||||
description: mcpTool.description || `MCP tool from ${serverName}`,
|
||||
schema: convertMcpSchemaToZod(mcpTool.inputSchema),
|
||||
func: async (params) => {
|
||||
// Call the MCP server
|
||||
const result = await mcpHub.callTool(serverName, mcpTool.name, params);
|
||||
|
||||
// Convert MCP response to expected format
|
||||
return {
|
||||
result: formatMcpResponse(result),
|
||||
};
|
||||
},
|
||||
};
|
||||
}
|
||||
```
|
||||
|
||||
### 2. Dynamic MCP Tool Registration
|
||||
|
||||
Register MCP tools when servers connect:
|
||||
|
||||
```typescript
|
||||
// In your MCP initialization code
|
||||
export async function registerMcpServerTools(serverName: string, mcpTools: McpTool[]) {
|
||||
const registry = ToolRegistry.getInstance();
|
||||
|
||||
for (const mcpTool of mcpTools) {
|
||||
registry.register({
|
||||
tool: createMcpToolWrapper(serverName, mcpTool),
|
||||
metadata: {
|
||||
id: `mcp_${serverName}_${mcpTool.name}`,
|
||||
displayName: mcpTool.displayName || mcpTool.name,
|
||||
description: mcpTool.description || `MCP tool from ${serverName}`,
|
||||
category: "mcp",
|
||||
// MCP tools are user-configurable by default
|
||||
isAlwaysEnabled: false,
|
||||
// Add any MCP-specific prompt instructions
|
||||
customPromptInstructions: mcpTool.systemPrompt,
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// When MCP server disconnects
|
||||
export function unregisterMcpServerTools(serverName: string) {
|
||||
const registry = ToolRegistry.getInstance();
|
||||
const allTools = registry.getAllTools();
|
||||
|
||||
// Remove tools from this server
|
||||
const toolsToKeep = allTools.filter((t) => !t.metadata.id.startsWith(`mcp_${serverName}_`));
|
||||
|
||||
registry.clear();
|
||||
registry.registerAll(toolsToKeep);
|
||||
|
||||
// Re-initialize built-in tools
|
||||
initializeBuiltinTools(app.vault);
|
||||
}
|
||||
```
|
||||
|
||||
### 3. Schema Conversion Helper
|
||||
|
||||
Convert MCP JSON Schema to Zod schema:
|
||||
|
||||
```typescript
|
||||
function convertMcpSchemaToZod(jsonSchema: any): z.ZodSchema {
|
||||
// Basic implementation - extend as needed
|
||||
if (jsonSchema.type === "object") {
|
||||
const shape: any = {};
|
||||
|
||||
for (const [key, prop] of Object.entries(jsonSchema.properties || {})) {
|
||||
const propSchema = prop as any;
|
||||
|
||||
if (propSchema.type === "string") {
|
||||
shape[key] = z.string();
|
||||
if (propSchema.description) {
|
||||
shape[key] = shape[key].describe(propSchema.description);
|
||||
}
|
||||
} else if (propSchema.type === "number") {
|
||||
shape[key] = z.number();
|
||||
} else if (propSchema.type === "boolean") {
|
||||
shape[key] = z.boolean();
|
||||
} else if (propSchema.type === "array") {
|
||||
shape[key] = z.array(z.any()); // Simplification
|
||||
} else if (propSchema.type === "object") {
|
||||
shape[key] = z.object({});
|
||||
}
|
||||
|
||||
// Handle optional properties
|
||||
if (!jsonSchema.required?.includes(key)) {
|
||||
shape[key] = shape[key].optional();
|
||||
}
|
||||
}
|
||||
|
||||
return z.object(shape);
|
||||
}
|
||||
|
||||
// Fallback for other types
|
||||
return z.any();
|
||||
}
|
||||
```
|
||||
|
||||
### 4. Settings Storage for MCP Tools
|
||||
|
||||
MCP tool preferences are stored in the same array as built-in tools:
|
||||
|
||||
```typescript
|
||||
// When enabling/disabling MCP tools:
|
||||
function updateMcpToolSetting(toolId: string, enabled: boolean) {
|
||||
const settings = getSettings();
|
||||
const enabledIds = new Set(settings.autonomousAgentEnabledToolIds || []);
|
||||
|
||||
if (enabled) {
|
||||
enabledIds.add(toolId);
|
||||
} else {
|
||||
enabledIds.delete(toolId);
|
||||
}
|
||||
|
||||
updateSetting("autonomousAgentEnabledToolIds", Array.from(enabledIds));
|
||||
}
|
||||
```
|
||||
|
||||
## How the System Works
|
||||
|
||||
### Tool Discovery Flow
|
||||
|
||||
1. **Initialization**: `initializeBuiltinTools()` registers all built-in tools
|
||||
2. **MCP Connection**: When MCP servers connect, their tools are dynamically registered
|
||||
3. **Settings UI**: `ToolSettingsSection` component reads from the registry to generate UI
|
||||
4. **Tool Execution**: `AutonomousAgentChainRunner.getAvailableTools()` filters tools based on settings
|
||||
|
||||
## Tool Call Rendering Roots
|
||||
|
||||
React invariant #409 surfaced when tool-call banners attempted to render into React roots that had already been unmounted. To prevent this regression, the plugin routes all banner rendering through the shared manager in `src/components/chat-components/toolCallRootManager.tsx`.
|
||||
|
||||
### Manager Responsibilities
|
||||
|
||||
- Tracks `{ root, isUnmounting }` per message/tool call via `window.__copilotToolCallRoots`.
|
||||
- `ensureToolCallRoot` finalises pending disposals and creates a new `createRoot` when needed.
|
||||
- `renderToolCallBanner` renders `<ToolCallBanner />` into the managed root; components never call `root.render` directly.
|
||||
- `removeToolCallRoot` and `cleanupMessageToolCallRoots` schedule unmounts on the next tick and drop entries only after disposal completes.
|
||||
- `cleanupStaleToolCallRoots` purges message IDs older than one hour to avoid leaking historical roots.
|
||||
|
||||
### Integration Notes
|
||||
|
||||
`ChatSingleMessage` keeps `const rootsRef = useRef(getMessageToolCallRoots(messageId))`, which provides a stable registry for each message. The component delegates all lifecycle calls to the manager and snapshots `rootsRef.current` inside effect cleanup to satisfy `react-hooks/exhaustive-deps`.
|
||||
|
||||
### Verification
|
||||
|
||||
Run the focused test to cover the streaming behaviour and tool-call integration:
|
||||
|
||||
```
|
||||
npm test -- src/components/chat-components/ChatSingleMessage.test.tsx
|
||||
```
|
||||
583
designdocs/todo/ACP_DESIGN.md
Normal file
|
|
@ -0,0 +1,583 @@
|
|||
# ACP Integration Design (Final)
|
||||
|
||||
Status: Final draft for implementation
|
||||
Date: 2026-02-19
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Add ACP-based agents (Claude Code, Codex, OpenCode first) as a first-class interaction paradigm in Copilot.
|
||||
|
||||
Core direction:
|
||||
|
||||
- ACP is a parallel runtime path.
|
||||
- Existing LangChain-based chat modes remain intact.
|
||||
- Long-term architecture is optimized for ACP as the primary agent path.
|
||||
|
||||
## 2. Final Decisions
|
||||
|
||||
1. Use `InteractionMode` (`llm` vs `agent`) instead of adding `ChainType.ACP`.
|
||||
2. Keep ACP runtime isolated from LangChain model/tool/memory/envelope stacks.
|
||||
3. Support file read/write + permission in agent mode.
|
||||
4. OpenCode model switching is required when ACP session model capabilities are available.
|
||||
|
||||
## 3. What ACP Changes (and What It Does Not)
|
||||
|
||||
### 3.1 Bypassed in `agent` mode
|
||||
|
||||
- `ChainManager` / `ChainRunner` request pipeline
|
||||
- `ChatModelManager` and provider model selection
|
||||
- `ContextManager` L1-L5 envelope construction
|
||||
- `LayerToMessagesConverter`
|
||||
- LangChain memory sync (`MemoryManager`, `updateChatMemory`)
|
||||
- LangChain-native `ToolRegistry` tool planning/calling
|
||||
- `getAIResponse()` in `src/langchainStream.ts`
|
||||
|
||||
### 3.2 Reused in `agent` mode
|
||||
|
||||
- Chat shell UI and message list containers
|
||||
- `MessageRepository` for display storage
|
||||
- `ChatUIState` subscription model
|
||||
- `ChatManager` as orchestration hub (with an ACP-specific send path)
|
||||
- Existing settings infrastructure
|
||||
|
||||
## 4. Runtime Architecture
|
||||
|
||||
### 4.1 Interaction Mode
|
||||
|
||||
Add top-level interaction mode in `src/aiParams.ts`:
|
||||
|
||||
```ts
|
||||
type InteractionMode = "llm" | "agent";
|
||||
```
|
||||
|
||||
- `llm` mode: current Copilot behavior unchanged.
|
||||
- `agent` mode: ACP pipeline and controls.
|
||||
|
||||
### 4.2 ACP Runtime Modules
|
||||
|
||||
Create a dedicated ACP namespace (`src/acp/`):
|
||||
|
||||
- `src/acp/ports/agent-client.port.ts` — IAgentClient interface (main contract)
|
||||
- `src/acp/adapters/acp.adapter.ts` — process spawn, ACP handshake, ndJSON stream, session update routing, permission queue
|
||||
- `src/acp/adapters/terminal-manager.ts` — handles ACP terminal callbacks: `terminal/create`, `terminal/output`, `terminal/kill`, `terminal/wait_for_exit`
|
||||
- `src/acp/types/*` — domain types (AgentConfig, SessionUpdate, PromptContent, etc.)
|
||||
- `src/acp/session/ACPManager.ts` — adapter lifecycle, agent switching, config resolution
|
||||
- `src/acp/context/AcpPromptAssembler.ts` — builds ACP prompt content from user input + attached context
|
||||
- `src/acp/updates/AcpUpdateReducer.ts` — routes session update notifications to message state
|
||||
- `src/acp/components/*` — agent selector, model/mode selectors, tool/permission/terminal renderers
|
||||
|
||||
Design rules:
|
||||
|
||||
- ACP modules must not depend on LangChain runtime components.
|
||||
- All ACP protocol details isolated in `adapters/` layer; domain and UI layers use port interfaces only.
|
||||
|
||||
### 4.3 ChatManager Parallel Path
|
||||
|
||||
Keep existing `sendMessage()` unchanged (LLM mode).
|
||||
Add `sendAgentMessage()` for ACP mode:
|
||||
|
||||
1. Create/store user message (display text).
|
||||
2. Build ACP prompt content via `AcpPromptAssembler`.
|
||||
3. Delegate to ACP manager for session/prompt streaming.
|
||||
|
||||
## 5. Context Strategy in Agent Mode
|
||||
|
||||
ACP agents manage their own context windows and tools. Copilot should only provide turn-scoped context.
|
||||
|
||||
Agent-mode prompt policy:
|
||||
|
||||
- Include current user message.
|
||||
- Include current-turn attached context only (notes/web selections/images).
|
||||
- Do not build or inject L2 cumulative context library.
|
||||
- Do not inject Copilot L4 conversation strip.
|
||||
- Do not force Copilot system prompt by default.
|
||||
|
||||
Attached context conversion (handled by `AcpPromptAssembler`):
|
||||
|
||||
- @mentioned notes → ACP `Resource` blocks if agent supports `embeddedContext` capability, otherwise embedded as text in the prompt.
|
||||
- Images → ACP image content blocks if agent supports `image` capability.
|
||||
- Web selections → embedded as text content blocks.
|
||||
|
||||
Optional advanced setting (off by default):
|
||||
|
||||
- "Prepend Copilot custom system instructions in agent mode"
|
||||
|
||||
This avoids context duplication and keeps prompt behavior agent-native.
|
||||
|
||||
## 6. File Read/Write and Permission Model
|
||||
|
||||
This section reflects observed ACP integration patterns and the final Copilot design requirements.
|
||||
|
||||
### 6.1 Finding: Typical ACP File Operation Patterns
|
||||
|
||||
Observed in common ACP client flows:
|
||||
|
||||
- During ACP initialize, clients may advertise:
|
||||
- `fs.readTextFile = false`
|
||||
- `fs.writeTextFile = false`
|
||||
- and still provide rich editing UX through:
|
||||
- agent `tool_call` / `tool_call_update` events (including diff content),
|
||||
- ACP `session/request_permission`,
|
||||
- permission UI actions mapped back to ACP permission responses.
|
||||
|
||||
Meaning:
|
||||
|
||||
- file editing permission flow is often agent-tool-based, not strictly dependent on ACP `fs/*` callbacks.
|
||||
|
||||
### 6.2 Copilot requirement and final design
|
||||
|
||||
Copilot agent mode must support file read/write with permission controls.
|
||||
|
||||
We support this through two channels:
|
||||
|
||||
1. Agent-tool permission channel (MVP required)
|
||||
|
||||
- Handle and render `tool_call`/`tool_call_update` diffs and statuses.
|
||||
- Handle `session/request_permission` with explicit approve/reject UI.
|
||||
- Default secure posture: no auto-allow by default.
|
||||
|
||||
2. ACP fs callback channel (compatibility extension)
|
||||
|
||||
- Implement real `fs/readTextFile` and `fs/writeTextFile` handlers against Obsidian vault APIs.
|
||||
- Gate writes behind the same explicit permission workflow.
|
||||
- Keep capability flags accurate per actual implementation.
|
||||
|
||||
Rationale:
|
||||
|
||||
- Channel 1 matches real behavior of target agents and is required immediately.
|
||||
- Channel 2 improves compatibility for agents that use ACP fs APIs directly.
|
||||
|
||||
### 6.3 Tool Call Architecture in Agent Mode
|
||||
|
||||
In `agent` mode, there are two tool lanes:
|
||||
|
||||
1. ACP-native agent tools (primary lane)
|
||||
|
||||
- Agent executes tools internally.
|
||||
- Agent emits `tool_call` / `tool_call_update`.
|
||||
- Copilot renders tool blocks and permission state.
|
||||
|
||||
2. Agent Skills (context-based, agent-executed)
|
||||
|
||||
- Skills are **markdown files and scripts** in a user-configured folder, similar to Claude Code's skill system.
|
||||
- Copilot surfaces skill files to the agent as context. The **agent** reads and executes them — Copilot does not run tools client-side.
|
||||
- Progressive disclosure: not all skills are dumped at once. Relevant skills are surfaced based on conversation context.
|
||||
|
||||
#### 6.3.1 Agent Skills Design
|
||||
|
||||
**What a skill is:**
|
||||
|
||||
- A `.md` file containing instructions, templates, or domain knowledge.
|
||||
- Optionally accompanied by scripts (shell, python, etc.) that the agent can execute via its own tool system.
|
||||
- Organized in a configurable skills folder within the vault (e.g., `copilot-skills/`).
|
||||
|
||||
**Example skill structure:**
|
||||
|
||||
```
|
||||
copilot-skills/
|
||||
├── vault-search.md # How to use miyo CLI/MCP for hybrid vault search
|
||||
├── web-search.md # Self-hosted web search endpoint and usage
|
||||
├── youtube-transcription.md # Self-hosted YouTube transcription service
|
||||
├── code-review.md # Instructions for how to review code in this project
|
||||
├── commit-conventions.md # Commit message format and rules
|
||||
├── vault-organization.md # How notes are structured in this vault
|
||||
├── scripts/
|
||||
│ ├── run-tests.sh # Test runner the agent can invoke
|
||||
│ └── lint-check.sh # Linting script
|
||||
└── templates/
|
||||
└── meeting-note.md # Template the agent can use when creating notes
|
||||
```
|
||||
|
||||
**Example: vault-search.md (miyo integration)**
|
||||
|
||||
```markdown
|
||||
# Vault Search
|
||||
|
||||
Use miyo for hybrid (semantic + keyword) search over this vault.
|
||||
|
||||
## CLI
|
||||
|
||||
miyo search "<query>" --limit 10
|
||||
|
||||
## MCP
|
||||
|
||||
miyo is also available as an MCP server for structured tool access.
|
||||
```
|
||||
|
||||
**Example: web-search.md (self-hosted service)**
|
||||
|
||||
```markdown
|
||||
# Web Search
|
||||
|
||||
Self-hosted web search via Firecrawl.
|
||||
|
||||
Endpoint: http://localhost:3002/v1/search
|
||||
Method: POST
|
||||
Body: {"query": "...", "limit": 5}
|
||||
Returns: JSON array of {title, url, content}
|
||||
```
|
||||
|
||||
This pattern covers all local tools and self-hosted services uniformly:
|
||||
|
||||
- **miyo** for vault hybrid search (CLI or MCP — agent chooses)
|
||||
- **Self-hosted Firecrawl** for web search
|
||||
- **Self-hosted Supadata** for YouTube transcription
|
||||
- Any future local service the user runs
|
||||
|
||||
Copilot never needs to bridge or proxy these. The agent calls them directly.
|
||||
|
||||
**How skills reach the agent:**
|
||||
|
||||
- Since working directory = vault root, the agent has filesystem access to skill files directly.
|
||||
- Copilot tells the agent about the skills folder location in the prompt context.
|
||||
- Copilot can optionally list available skill filenames so the agent knows what's there.
|
||||
- The agent decides which skills to read and follow — full agency stays with the agent.
|
||||
|
||||
**Progressive disclosure strategy:**
|
||||
|
||||
- Level 1: Always include a skill index (list of filenames + one-line descriptions) in the prompt.
|
||||
- Level 2: Include full content of skills tagged as "always active" (e.g., project conventions).
|
||||
- Level 3: Agent reads additional skill files on demand via its own file read tools.
|
||||
|
||||
**Skill management UI:**
|
||||
|
||||
- Settings: configure skills folder path.
|
||||
- Skills browser: list discovered skills, toggle "always active" flag per skill.
|
||||
- No client-side execution — Copilot never runs skill scripts. The agent does.
|
||||
|
||||
**Key principle:**
|
||||
|
||||
- Skills are context, not tools. Copilot provides them. The agent acts on them.
|
||||
- This keeps the ACP path clean — no client-side tool execution, no LangChain coupling.
|
||||
- Analogous to how Claude Code reads CLAUDE.md files for project-specific instructions.
|
||||
|
||||
## 7. Agent and Session State
|
||||
|
||||
### 7.1 Agent presets
|
||||
|
||||
Built-in defaults:
|
||||
|
||||
- Claude Code
|
||||
- Codex
|
||||
- OpenCode
|
||||
|
||||
Each preset is user-editable:
|
||||
|
||||
- `id`, `displayName`, `command`, `args`, `env`.
|
||||
|
||||
Do not hardcode a single arg style across versions.
|
||||
|
||||
- OpenCode commonly uses an `acp` subcommand in examples.
|
||||
- Some agents may use flags such as `--acp`.
|
||||
- Presets are defaults, not strict assumptions.
|
||||
|
||||
### 7.2 Capabilities
|
||||
|
||||
Track capabilities from ACP initialize/new session:
|
||||
|
||||
- prompt capabilities (image/resource)
|
||||
- session capabilities (mode/model/list/load/resume/fork if available)
|
||||
|
||||
All UI controls are capability-gated.
|
||||
|
||||
### 7.3 OpenCode model switching
|
||||
|
||||
Requirement implementation:
|
||||
|
||||
- Use ACP session models from `newSession` / loaded session data.
|
||||
- Show model selector when multiple models are exposed.
|
||||
- Switch with `unstable_setSessionModel`.
|
||||
- Use optimistic UI update with rollback on error.
|
||||
|
||||
Important protocol note:
|
||||
|
||||
- `current_mode_update` is mode-specific.
|
||||
- Do not rely on it as model-change confirmation.
|
||||
|
||||
### 7.4 Error handling and reconnection
|
||||
|
||||
Agent process errors are surfaced explicitly in the chat UI. No silent recovery.
|
||||
|
||||
**Error categories:**
|
||||
|
||||
- **Spawn failure** (command not found, permission denied): Show error with setup guidance. Detect via exit code 127, `ENOENT`.
|
||||
- **Agent crash** (unexpected process exit): Show error in chat. Session state → "disconnected".
|
||||
- **Protocol error** (ACP JSON-RPC errors): Show error message from agent. Session remains connected if process alive.
|
||||
- **Silent failure** (agent returns empty response): Detect via zero session updates + `end_turn`. Check stderr for API key / auth hints.
|
||||
|
||||
**Reconnection strategy:**
|
||||
|
||||
- No automatic reconnection. Agent processes are stateful — blindly restarting loses session context.
|
||||
- User-initiated: "Restart Agent" action kills process, re-spawns, creates new session.
|
||||
- If agent supports `loadSession`: offer "Restart and Reload" to re-spawn + reload previous session history.
|
||||
- Display clear connection status indicator (connected / busy / disconnected / error) in ChatControls.
|
||||
|
||||
## 8. UI Behavior
|
||||
|
||||
### 8.1 ChatControls
|
||||
|
||||
When `interactionMode === "llm"`:
|
||||
|
||||
- keep existing chain/model controls.
|
||||
|
||||
When `interactionMode === "agent"`:
|
||||
|
||||
- show agent selector.
|
||||
- show mode selector when ACP modes available.
|
||||
- show model selector when ACP models available.
|
||||
- show session/connection status indicator.
|
||||
- hide Copilot Plus LangChain tool toggles and command injection controls.
|
||||
- show skills folder indicator (configured/not configured, skill count).
|
||||
|
||||
### 8.2 Message rendering in agent mode
|
||||
|
||||
Render ACP updates as first-class content:
|
||||
|
||||
- assistant text stream (`agent_message_chunk`)
|
||||
- thought stream (`agent_thought_chunk`)
|
||||
- tool call blocks (`tool_call`, `tool_call_update`)
|
||||
- permission controls (`request_permission`)
|
||||
- terminal output blocks
|
||||
- plan blocks
|
||||
|
||||
Do not reuse legacy marker-based agent rendering as the primary ACP path.
|
||||
|
||||
Additionally show skill context metadata when skills are active:
|
||||
|
||||
- skills folder path indicator
|
||||
- list of "always active" skills included in prompt context
|
||||
|
||||
## 9. Message Flow
|
||||
|
||||
### 9.1 LLM mode (unchanged)
|
||||
|
||||
Existing path remains as-is:
|
||||
|
||||
- `Chat` -> `ChatUIState` -> `ChatManager.sendMessage()` -> LangChain flow.
|
||||
|
||||
### 9.2 Agent mode
|
||||
|
||||
New path:
|
||||
|
||||
1. `Chat` checks `interactionMode === "agent"`.
|
||||
2. `ChatManager.sendAgentMessage()` stores user message and builds ACP prompt content.
|
||||
3. ACP manager ensures process/session and sends `session/prompt`.
|
||||
4. Session updates stream back into message state.
|
||||
5. Cancel maps to ACP `session/cancel`.
|
||||
|
||||
## 10. Settings Additions
|
||||
|
||||
Extend `CopilotSettings` with ACP section:
|
||||
|
||||
- `interactionModeDefault` (optional)
|
||||
- `acpDefaultAgentId`
|
||||
- `acpAgents: ACPAgentConfig[]`
|
||||
- `acpAutoAllowPermissions` (default false)
|
||||
- `acpSkillsFolderPath` (default `"copilot-skills"`, vault-relative)
|
||||
- optional ACP diagnostics/logging toggles
|
||||
|
||||
## 11. Implementation Plan
|
||||
|
||||
### Phase 1: Core ACP lane
|
||||
|
||||
**Goal**: End-to-end text streaming with a single agent (Claude Code).
|
||||
|
||||
**New files:**
|
||||
| File | Ported from reference | Description |
|
||||
|---|---|---|
|
||||
| `src/acp/ports/agent-client.port.ts` | `domain/ports/agent-client.port.ts` | IAgentClient interface |
|
||||
| `src/acp/types/agentConfig.ts` | `domain/models/agent-config.ts` | AgentConfig, BaseAgentSettings |
|
||||
| `src/acp/types/sessionUpdate.ts` | `domain/models/session-update.ts` | SessionUpdate union type |
|
||||
| `src/acp/types/promptContent.ts` | `domain/models/prompt-content.ts` | PromptContent types |
|
||||
| `src/acp/types/sessionState.ts` | `domain/models/chat-session.ts` | Mode/model state types |
|
||||
| `src/acp/types/agentError.ts` | `domain/models/agent-error.ts` | Error types |
|
||||
| `src/acp/adapters/acp.adapter.ts` | `adapters/acp/acp.adapter.ts` | Core ACP adapter (~1200 lines) |
|
||||
| `src/acp/adapters/acp-type-converter.ts` | `adapters/acp/acp-type-converter.ts` | Domain ↔ SDK types |
|
||||
| `src/acp/utils/shellUtils.ts` | `shared/shell-utils.ts` | Login shell wrapping |
|
||||
| `src/acp/utils/errorUtils.ts` | `shared/acp-error-utils.ts` | Error parsing |
|
||||
| `src/acp/session/ACPManager.ts` | new | Adapter lifecycle singleton |
|
||||
| `src/acp/components/AgentSelector.tsx` | new | Agent picker dropdown |
|
||||
|
||||
**Modified files:**
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `src/aiParams.ts` | Add `InteractionMode` type + Jotai atoms |
|
||||
| `src/core/ChatManager.ts` | Add `sendAgentMessage()` method |
|
||||
| `src/components/Chat.tsx` | Branch `handleSendMessage` on interaction mode |
|
||||
| `src/components/chat-components/ChatControls.tsx` | Add mode toggle + agent selector |
|
||||
| `src/settings/model.ts` | Add ACP settings fields |
|
||||
| `package.json` | Add `@agentclientprotocol/sdk` dependency |
|
||||
|
||||
**Workload**: Largest phase. ~15 new files, ~5 modified files. The ACP adapter is the heaviest piece (~1200 lines ported from reference, adapted for Copilot patterns). Shell utils and type definitions are mostly direct ports. ACPManager and ChatManager integration are new code.
|
||||
|
||||
**Exit criteria**: Select Claude Code in agent mode → type message → see streaming text response in chat.
|
||||
|
||||
---
|
||||
|
||||
### Phase 2: Permissions + tool call rendering
|
||||
|
||||
**Goal**: Full tool call UX — diffs, terminal output, permission prompts.
|
||||
|
||||
**New files:**
|
||||
| File | Ported from reference | Description |
|
||||
|---|---|---|
|
||||
| `src/acp/adapters/terminal-manager.ts` | `shared/terminal-manager.ts` | Terminal lifecycle + output buffering |
|
||||
| `src/acp/components/ToolCallBlock.tsx` | new (reference has `ToolCallRenderer.tsx`) | Tool call status, kind, title, locations |
|
||||
| `src/acp/components/DiffViewer.tsx` | new (reference has `DiffBlock.tsx`) | File diff rendering |
|
||||
| `src/acp/components/TerminalOutput.tsx` | new (reference has `TerminalBlock.tsx`) | Terminal command output |
|
||||
| `src/acp/components/PermissionRequestUI.tsx` | new (reference has `PermissionRequestSection.tsx`) | Approve/deny inline buttons |
|
||||
| `src/acp/components/PlanBlock.tsx` | new (reference has `PlanBlock.tsx`) | Execution plan task list |
|
||||
| `src/acp/updates/AcpUpdateReducer.ts` | new | Routes session updates to message content |
|
||||
|
||||
**Modified files:**
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `src/components/chat-components/ChatMessages.tsx` | Detect and render ACP content types |
|
||||
| `src/settings/model.ts` | Add `acpAutoAllowPermissions` |
|
||||
|
||||
**Workload**: Medium-heavy. ~7 new component files. TerminalManager is substantial (~500 lines from reference). UI components are mostly new but follow patterns from reference plugin. AcpUpdateReducer is new routing logic.
|
||||
|
||||
**Exit criteria**: Agent performs file edit → see diff in chat → permission prompt appears → approve → agent continues. Terminal commands show output blocks.
|
||||
|
||||
---
|
||||
|
||||
### Phase 3: Mode/model controls + settings
|
||||
|
||||
**Goal**: Full agent configuration, OpenCode model switching, mode selection.
|
||||
|
||||
**New files:**
|
||||
| File | Description |
|
||||
|---|---|
|
||||
| `src/acp/components/AgentModelSelector.tsx` | Model dropdown populated from ACP session models |
|
||||
| `src/acp/components/AgentModeSelector.tsx` | Mode dropdown populated from ACP session modes |
|
||||
| `src/acp/components/AgentSettingsTab.tsx` | Full settings UI for agent configuration |
|
||||
|
||||
**Modified files:**
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `src/acp/adapters/acp.adapter.ts` | Add `setSessionMode()`, `setSessionModel()` calls |
|
||||
| `src/components/chat-components/ChatControls.tsx` | Wire model/mode selectors, connection status |
|
||||
| `src/settings/model.ts` | Add built-in agent presets, custom agent support |
|
||||
|
||||
**Workload**: Medium. 3 new UI components, moderate adapter additions. Settings tab is the largest piece — per-agent command/args/env/apikey editing. Model/mode selectors are small but need optimistic UI + rollback.
|
||||
|
||||
**Exit criteria**: Select OpenCode → see model dropdown → switch to minimax-2.5 → agent confirms model change. Edit Claude Code command path in settings → reconnect works.
|
||||
|
||||
---
|
||||
|
||||
### Phase 4: Agent Skills
|
||||
|
||||
**Goal**: Skills folder discovery, index generation, always-active injection, skills browser.
|
||||
|
||||
**New files:**
|
||||
| File | Description |
|
||||
|---|---|
|
||||
| `src/acp/context/AcpPromptAssembler.ts` | Builds prompt content: user text + mentions + skill index + active skills |
|
||||
| `src/acp/context/skillDiscovery.ts` | Scans skills folder, extracts index (filename + first-line description) |
|
||||
| `src/acp/components/SkillsBrowser.tsx` | List skills, toggle "always active" per skill |
|
||||
|
||||
**Modified files:**
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `src/acp/session/ACPManager.ts` | Inject skill context into prompts |
|
||||
| `src/settings/model.ts` | Add `acpSkillsFolderPath` setting |
|
||||
|
||||
**Workload**: Medium-light. Skill discovery is straightforward filesystem scanning. AcpPromptAssembler assembles the prompt with skill index + active skill content. Browser UI is a simple list with toggles.
|
||||
|
||||
**Exit criteria**: Configure `copilot-skills/` folder → skills appear in browser → mark `vault-search.md` as always-active → send message → agent sees skill index + vault-search.md content in prompt → agent uses miyo to search.
|
||||
|
||||
---
|
||||
|
||||
### Phase 5: ACP fs callback compatibility
|
||||
|
||||
**Goal**: Real vault read/write via ACP `fs/*` callbacks for agents that use them.
|
||||
|
||||
**New files:**
|
||||
| File | Ported from reference | Description |
|
||||
|---|---|---|
|
||||
| `src/acp/adapters/vault-adapter.ts` | `adapters/obsidian/vault.adapter.ts` | Bridges ACP fs to Obsidian vault API |
|
||||
|
||||
**Modified files:**
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `src/acp/adapters/acp.adapter.ts` | Enable `fs.readTextFile`/`fs.writeTextFile` capabilities, wire to vault adapter |
|
||||
|
||||
**Workload**: Light. Single adapter file. Read is simple vault file read. Write needs permission prompt before executing (reuse permission UI from Phase 2).
|
||||
|
||||
**Exit criteria**: Agent using ACP fs API reads vault file → gets content. Agent writes file → permission prompt → approve → file written to vault.
|
||||
|
||||
---
|
||||
|
||||
### Phase 6: Session lifecycle enhancements
|
||||
|
||||
**Goal**: Session persistence, load/resume, session history browser.
|
||||
|
||||
**New files:**
|
||||
| File | Description |
|
||||
|---|---|
|
||||
| `src/acp/components/SessionHistoryModal.tsx` | Session list + load/resume actions |
|
||||
|
||||
**Modified files:**
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `src/acp/adapters/acp.adapter.ts` | Add `listSessions()`, `loadSession()`, `resumeSession()`, `forkSession()` |
|
||||
| `src/acp/session/ACPManager.ts` | Session metadata persistence, capability-gated session operations |
|
||||
| `src/settings/model.ts` | Add saved session metadata storage |
|
||||
|
||||
**Workload**: Medium. Session list/load requires adapter additions and UI. All session features are capability-gated — only show UI if agent supports them. Fork is unstable and lowest priority within this phase.
|
||||
|
||||
**Exit criteria**: Close and reopen Obsidian → switch to agent mode → "Load Session" shows previous sessions → select one → conversation history replays → can continue conversation.
|
||||
|
||||
## 12. Risks and Mitigations
|
||||
|
||||
- ACP unstable methods: isolate protocol details in adapter layer.
|
||||
- Process lifecycle errors: explicit reconnect and error surfacing (see 7.4).
|
||||
- Permission safety: default deny/explicit user action.
|
||||
- Context bloat: strict per-turn context assembly only.
|
||||
- UI complexity: mode-gated controls and capability-based rendering.
|
||||
|
||||
## 13. Future Decisions (Post-MVP)
|
||||
|
||||
The following items are intentionally deferred from MVP and should be revisited in later phases:
|
||||
|
||||
1. `@` command behavior in agent mode
|
||||
|
||||
- Decide whether `@vault/@websearch/@composer/@memory` remain plain text only, or trigger skill-hint injection behavior.
|
||||
|
||||
2. Skill file schema contract
|
||||
|
||||
- Decide whether skills require frontmatter metadata (`name`, `description`, `alwaysActive`, `tags`) vs free-form markdown.
|
||||
|
||||
3. Progressive disclosure algorithm
|
||||
|
||||
- Define how skill relevance is selected (keyword, tags, manual pinning, hybrid scoring).
|
||||
|
||||
4. Skill prompt budget limits
|
||||
|
||||
- Set hard caps for skill index size and total always-active skill content injected per turn.
|
||||
|
||||
5. Tool integration boundary
|
||||
|
||||
- Confirm whether agent mode stays context-only for skills, or eventually allows selective client-side Copilot tool execution.
|
||||
|
||||
6. Script safety model
|
||||
|
||||
- Define Copilot-level safeguards for skill scripts (trusted folders, warnings, policy prompts) in addition to agent permission flow.
|
||||
|
||||
7. Skills scope model
|
||||
|
||||
- Decide global vault-wide skills vs project/profile-specific skill sets.
|
||||
|
||||
8. Skills folder constraints
|
||||
|
||||
- Decide vault-relative only skills folders vs external/absolute path support.
|
||||
|
||||
## 14. Acceptance Criteria
|
||||
|
||||
- Agent mode can run Claude Code, Codex, OpenCode.
|
||||
- OpenCode model switching works when the agent exposes session models.
|
||||
- File operations are permission-controlled and visible in chat.
|
||||
- LLM modes are behaviorally unchanged.
|
||||
- ACP mode avoids LangChain context/model/tool/memory pipelines.
|
||||
- ACP mode never invokes LangChain autonomous tool planning/execution loops.
|
||||
- Agent Skills (md files + scripts) are surfaced as context; agent reads and executes them, not Copilot.
|
||||
239
designdocs/todo/AGENT_PLANNING_REFLECTION_V0.md
Normal file
|
|
@ -0,0 +1,239 @@
|
|||
# Agent Planning + Reflection Visibility (v0)
|
||||
|
||||
**Date:** 2026-02-10
|
||||
**Status:** Draft
|
||||
**Scope:** Autonomous Agent (`AutonomousAgentChainRunner`) only
|
||||
|
||||
## 1. Problem Statement
|
||||
|
||||
The current autonomous agent loop is functional and simple, but it has two gaps:
|
||||
|
||||
1. No explicit machine-readable plan state in the ReAct loop.
|
||||
2. Reasoning visibility is mostly tool-call/result summaries, with weak iteration-level reflection.
|
||||
|
||||
Today, planning is implicit in model text and tool order. The UI (`AgentReasoningBlock`) only sees serialized step strings, so users cannot clearly track "what is the current plan" vs "what just happened".
|
||||
|
||||
## 2. Current Baseline (What We Have)
|
||||
|
||||
- ReAct loop with native tool calling in `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts`.
|
||||
- Reasoning block state/serialization in `src/LLMProviders/chainRunner/utils/AgentReasoningState.ts`.
|
||||
- Reasoning UI rendering in `src/components/chat-components/AgentReasoningBlock.tsx` and parsing in `src/components/chat-components/ChatSingleMessage.tsx`.
|
||||
- Tool registry and metadata model in `src/tools/ToolRegistry.ts` and `src/tools/builtinTools.ts`.
|
||||
|
||||
This is already a solid base for a minimal planner because:
|
||||
|
||||
- The loop already supports iterative tool decisions.
|
||||
- The reasoning block already supports rolling vs full history.
|
||||
- Tools are already typed with Zod and routed through one registry.
|
||||
|
||||
## 3. Goals
|
||||
|
||||
1. Add a minimal planning primitive (`write_todos`) that fits the existing sequential ReAct loop.
|
||||
2. Improve per-iteration reasoning visibility without exposing chain-of-thought.
|
||||
3. Keep the implementation robust with minimal new state.
|
||||
4. Prepare a clean extension point for future subagents and context encapsulation.
|
||||
|
||||
## 4. Non-Goals (v0)
|
||||
|
||||
1. No multi-agent orchestration in this phase.
|
||||
2. No persistent cross-turn planner memory.
|
||||
3. No complex planner DAG or dependency graph.
|
||||
4. No major UI rewrite of Reasoning Block.
|
||||
|
||||
## 5. v0 Design Overview
|
||||
|
||||
### 5.1 Add a Minimal Planner Tool: `write_todos`
|
||||
|
||||
Introduce a lightweight built-in tool that updates the agent's execution checklist.
|
||||
|
||||
Tool semantics:
|
||||
|
||||
- Input is the full current todo snapshot (replace semantics, not patch semantics).
|
||||
- Output is a compact structured acknowledgement.
|
||||
- No file I/O, no vault mutation, no side effects outside the in-memory run state.
|
||||
|
||||
Example schema:
|
||||
|
||||
```ts
|
||||
const writeTodosSchema = z.object({
|
||||
todos: z
|
||||
.array(
|
||||
z.object({
|
||||
id: z.string().min(1).max(40),
|
||||
content: z.string().min(1).max(140),
|
||||
status: z.enum(["pending", "in_progress", "completed"]),
|
||||
})
|
||||
)
|
||||
.min(1)
|
||||
.max(8),
|
||||
focus: z.string().max(40).optional(),
|
||||
note: z.string().max(200).optional(),
|
||||
});
|
||||
```
|
||||
|
||||
Example result payload:
|
||||
|
||||
```json
|
||||
{
|
||||
"ok": true,
|
||||
"revision": 3,
|
||||
"todoCount": 4,
|
||||
"inProgress": "read_note_context"
|
||||
}
|
||||
```
|
||||
|
||||
Why replace semantics:
|
||||
|
||||
- Easier for model to reason about.
|
||||
- Deterministic state transitions.
|
||||
- No merge/conflict logic in runner.
|
||||
|
||||
### 5.2 ReAct Loop Integration (Minimal Changes)
|
||||
|
||||
In `runReActLoop`:
|
||||
|
||||
- Keep one loop and one tool execution path.
|
||||
- Special-case `write_todos` before normal tool execution.
|
||||
- Convert planner updates into reasoning events and a compact `ToolMessage` acknowledgement.
|
||||
|
||||
Pseudo-flow:
|
||||
|
||||
1. Model returns `tool_calls`.
|
||||
2. If call is `write_todos`, apply/update in-memory planner state.
|
||||
3. Emit reasoning step(s) with `[Plan]` prefix.
|
||||
4. Push tool result message so model can continue.
|
||||
5. Continue loop unchanged for normal tools.
|
||||
|
||||
Guardrails:
|
||||
|
||||
- Max 2 consecutive planner-only iterations.
|
||||
- If planner loops, return tool error: `"planner_overuse_execute_next_step"`.
|
||||
- If planner args invalid, return schema error and continue loop.
|
||||
|
||||
### 5.3 Better Reflection Visibility in Existing Reasoning Block
|
||||
|
||||
Keep the existing component but make steps more legible by phase-tagging events.
|
||||
|
||||
Step tags (string prefix only, no UI rewrite required):
|
||||
|
||||
- `[Plan]` todo updates and step ordering
|
||||
- `[Act]` tool call intent
|
||||
- `[Obs]` tool result summary
|
||||
- `[Reflect]` model's concise iteration reflection
|
||||
|
||||
Implementation detail:
|
||||
|
||||
- Reuse current `addReasoningStep` and `allReasoningSteps`.
|
||||
- Add small extraction helper for reflection text from `AIMessage.content` per iteration.
|
||||
- Enforce short reflection summaries (single sentence, capped length).
|
||||
|
||||
This gives better visibility immediately with minimal parser/rendering changes.
|
||||
|
||||
### 5.4 Prompting Updates
|
||||
|
||||
Add tool guidance for `write_todos` via tool metadata and agent prompt section.
|
||||
|
||||
Rules:
|
||||
|
||||
1. Use `write_todos` for multi-step tasks (>=2 meaningful actions).
|
||||
2. First planner call should happen before the first expensive external tool when task is non-trivial.
|
||||
3. Keep todos short and action-oriented.
|
||||
4. Update statuses as execution progresses.
|
||||
5. Do not repeatedly rewrite unchanged todos.
|
||||
|
||||
## 6. Data Model (v0 Sidecar State)
|
||||
|
||||
Add in-memory runtime state in `AutonomousAgentChainRunner`:
|
||||
|
||||
```ts
|
||||
interface PlannerState {
|
||||
revision: number;
|
||||
todos: Array<{ id: string; content: string; status: "pending" | "in_progress" | "completed" }>;
|
||||
focus?: string;
|
||||
updatedAt: number;
|
||||
}
|
||||
|
||||
interface ReasoningEvent {
|
||||
phase: "plan" | "act" | "obs" | "reflect";
|
||||
summary: string;
|
||||
iteration: number;
|
||||
timestamp: number;
|
||||
}
|
||||
```
|
||||
|
||||
No persistence changes are required for v0. Existing chat persistence already strips reasoning markers.
|
||||
|
||||
## 7. Extensibility Path: Context Capsule for Future Subagents
|
||||
|
||||
To support near-future subagents without redesigning the loop, add one abstraction now:
|
||||
|
||||
```ts
|
||||
interface ContextCapsule {
|
||||
goal: string;
|
||||
planSnapshot?: PlannerState;
|
||||
keyFindings: string[];
|
||||
artifacts: Array<{ type: string; ref: string; summary: string }>;
|
||||
nextActions?: string[];
|
||||
}
|
||||
```
|
||||
|
||||
v0 usage:
|
||||
|
||||
- Single agent creates this in-memory as a byproduct (optional, debug-only).
|
||||
|
||||
Future subagent usage:
|
||||
|
||||
- Parent agent passes a scoped goal.
|
||||
- Subagent returns only a compact `ContextCapsule` (not full transcript).
|
||||
- Parent injects capsule summary into next decision turn as tool result.
|
||||
|
||||
This keeps context encapsulated and token usage bounded.
|
||||
|
||||
## 8. Minimal File-Level Change Plan
|
||||
|
||||
1. Add `src/tools/PlannerTools.ts` with `write_todos` tool.
|
||||
2. Register tool in `src/tools/builtinTools.ts`.
|
||||
3. Update `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts`:
|
||||
- planner sidecar state
|
||||
- special handling for `write_todos`
|
||||
- tagged reasoning events (`[Plan]/[Act]/[Obs]/[Reflect]`)
|
||||
4. Optional small helper updates in `src/LLMProviders/chainRunner/utils/AgentReasoningState.ts` for reflection extraction formatting.
|
||||
5. Add tests:
|
||||
- planner tool schema/validation
|
||||
- loop behavior with planner-only + mixed tool calls
|
||||
- reasoning step tagging regression
|
||||
|
||||
## 9. Acceptance Criteria
|
||||
|
||||
1. Complex user query shows at least one `[Plan]`, one `[Act]`, and one `[Obs]` in reasoning steps.
|
||||
2. Planner updates do not break normal ReAct completion behavior.
|
||||
3. Agent still terminates on timeout/max-iterations as before.
|
||||
4. No regression in non-planner queries.
|
||||
|
||||
## 10. Risks and Mitigations
|
||||
|
||||
1. Model ignores planner tool:
|
||||
- Mitigation: planner is optional; loop still works exactly as today.
|
||||
2. Planner spam:
|
||||
- Mitigation: cap consecutive planner-only iterations.
|
||||
3. Token bloat from verbose todos:
|
||||
- Mitigation: hard limits on item count and text length.
|
||||
4. Over-exposure of hidden reasoning:
|
||||
- Mitigation: only allow concise operational reflection summaries.
|
||||
|
||||
## 11. Rollout
|
||||
|
||||
1. Ship behind a feature flag (e.g., `enableAgentPlannerV0`).
|
||||
2. Enable for internal testing first.
|
||||
3. Validate on representative flows: search-heavy, note reading, and composer edit tasks.
|
||||
4. Enable by default after stability pass.
|
||||
|
||||
## 12. Open Questions
|
||||
|
||||
1. Should `write_todos` be always enabled or user-configurable?
|
||||
2. Should planner state be exposed in any UI beyond Reasoning Block?
|
||||
3. Should we persist final plan snapshot in message metadata for debugging?
|
||||
|
||||
---
|
||||
|
||||
This v0 keeps the architecture simple: one sequential ReAct loop, one lightweight planning tool, and better reasoning visibility now, while setting up a clean context-capsule path for subagents later.
|
||||
572
designdocs/todo/AGENT_REASONING_BLOCK.md
Normal file
|
|
@ -0,0 +1,572 @@
|
|||
# Agent Reasoning Block Implementation Plan
|
||||
|
||||
## Overview
|
||||
|
||||
Replace the current tool call banner with a new **Agent Reasoning Block** - a multi-line collapsible block that shows the agent's reasoning process during execution, then collapses to "Thought for N s" during final response streaming.
|
||||
|
||||
---
|
||||
|
||||
## Design Reference
|
||||
|
||||
**During Agent Loop (Expanded):**
|
||||
|
||||
```
|
||||
┌──────────────────────────────────────────────────┐
|
||||
│ ⠿ Reasoning · 9s │
|
||||
│ │
|
||||
│ • Searching notes for "machine learning" │
|
||||
│ • Found 5 relevant notes, analyzing content │
|
||||
└──────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
**During Final Response (Collapsed):**
|
||||
|
||||
```
|
||||
┌──────────────────────────────────────────────────┐
|
||||
│ ▸ Thought for 12s │
|
||||
└──────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
**After Response Complete (Expandable):**
|
||||
|
||||
```
|
||||
┌──────────────────────────────────────────────────┐
|
||||
│ ▸ Thought for 12s [click to expand] │
|
||||
└──────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Naming
|
||||
|
||||
| Term | Description |
|
||||
| ------------------------- | -------------------------------------------- |
|
||||
| **Agent Reasoning Block** | The full UI component |
|
||||
| **Reasoning Steps** | Individual bullet points (1-2 per iteration) |
|
||||
| **Reasoning Timer** | Elapsed seconds counter |
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
### State Machine
|
||||
|
||||
```
|
||||
┌─────────────┐ tool call ┌─────────────┐
|
||||
│ IDLE │ ────────────────▶ │ REASONING │
|
||||
└─────────────┘ └─────────────┘
|
||||
│
|
||||
│ final response starts
|
||||
▼
|
||||
┌─────────────┐
|
||||
│ COLLAPSED │
|
||||
└─────────────┘
|
||||
│
|
||||
│ response complete
|
||||
▼
|
||||
┌─────────────┐
|
||||
│ COMPLETE │
|
||||
└─────────────┘
|
||||
```
|
||||
|
||||
### Data Flow
|
||||
|
||||
```
|
||||
AutonomousAgentChainRunner
|
||||
│
|
||||
├── onReasoningStart(timestamp)
|
||||
│ └── Start timer, set state = REASONING
|
||||
│
|
||||
├── onReasoningStep(summary: string)
|
||||
│ └── Add bullet point to steps array
|
||||
│
|
||||
├── onReasoningEnd()
|
||||
│ └── Set state = COLLAPSED, stop timer
|
||||
│
|
||||
└── Final streaming via updateCurrentAiMessage
|
||||
└── Normal text streaming (block stays collapsed)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Implementation Phases
|
||||
|
||||
### Phase 1: Data Model & State
|
||||
|
||||
**File:** `src/LLMProviders/chainRunner/utils/AgentReasoningState.ts` (new)
|
||||
|
||||
```typescript
|
||||
export interface ReasoningStep {
|
||||
timestamp: number;
|
||||
summary: string; // e.g., "Searching notes for 'AI'"
|
||||
toolName?: string;
|
||||
}
|
||||
|
||||
export interface AgentReasoningState {
|
||||
status: "idle" | "reasoning" | "collapsed" | "complete";
|
||||
startTime: number | null;
|
||||
elapsedSeconds: number;
|
||||
steps: ReasoningStep[];
|
||||
}
|
||||
|
||||
export function createInitialReasoningState(): AgentReasoningState {
|
||||
return {
|
||||
status: "idle",
|
||||
startTime: null,
|
||||
elapsedSeconds: 0,
|
||||
steps: [],
|
||||
};
|
||||
}
|
||||
|
||||
// Serialize to marker format (embedded in message)
|
||||
export function serializeReasoningBlock(state: AgentReasoningState): string {
|
||||
const data = {
|
||||
elapsed: state.elapsedSeconds,
|
||||
steps: state.steps.map((s) => s.summary),
|
||||
};
|
||||
return `<!--REASONING_BLOCK:${JSON.stringify(data)}-->`;
|
||||
}
|
||||
|
||||
// Parse from marker format
|
||||
export function parseReasoningBlock(marker: string): { elapsed: number; steps: string[] } | null {
|
||||
const match = marker.match(/<!--REASONING_BLOCK:(.+?)-->/);
|
||||
if (!match) return null;
|
||||
try {
|
||||
return JSON.parse(match[1]);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Phase 2: Update AutonomousAgentChainRunner
|
||||
|
||||
**File:** `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts`
|
||||
|
||||
**Changes:**
|
||||
|
||||
1. Add reasoning state tracking:
|
||||
|
||||
```typescript
|
||||
private reasoningState: AgentReasoningState = createInitialReasoningState();
|
||||
private reasoningTimerInterval: NodeJS.Timeout | null = null;
|
||||
```
|
||||
|
||||
2. Add helper methods:
|
||||
|
||||
```typescript
|
||||
private startReasoningTimer(updateFn: (message: string) => void): void {
|
||||
this.reasoningState = {
|
||||
status: 'reasoning',
|
||||
startTime: Date.now(),
|
||||
elapsedSeconds: 0,
|
||||
steps: [],
|
||||
};
|
||||
|
||||
// Update every 100ms for responsive timer
|
||||
this.reasoningTimerInterval = setInterval(() => {
|
||||
if (this.reasoningState.startTime) {
|
||||
this.reasoningState.elapsedSeconds = Math.floor(
|
||||
(Date.now() - this.reasoningState.startTime) / 1000
|
||||
);
|
||||
// Emit updated reasoning block
|
||||
updateFn(this.buildReasoningBlockMarkup());
|
||||
}
|
||||
}, 100);
|
||||
}
|
||||
|
||||
private addReasoningStep(summary: string): void {
|
||||
// Keep only last 2 steps for concise display
|
||||
this.reasoningState.steps.push({
|
||||
timestamp: Date.now(),
|
||||
summary,
|
||||
});
|
||||
if (this.reasoningState.steps.length > 2) {
|
||||
this.reasoningState.steps.shift();
|
||||
}
|
||||
}
|
||||
|
||||
private stopReasoningTimer(): void {
|
||||
if (this.reasoningTimerInterval) {
|
||||
clearInterval(this.reasoningTimerInterval);
|
||||
this.reasoningTimerInterval = null;
|
||||
}
|
||||
this.reasoningState.status = 'collapsed';
|
||||
}
|
||||
|
||||
private buildReasoningBlockMarkup(): string {
|
||||
const { status, elapsedSeconds, steps } = this.reasoningState;
|
||||
|
||||
if (status === 'idle') return '';
|
||||
|
||||
// Use special marker that ChatSingleMessage will parse
|
||||
const stepsJson = JSON.stringify(steps.map(s => s.summary));
|
||||
return `<!--AGENT_REASONING:${status}:${elapsedSeconds}:${stepsJson}-->`;
|
||||
}
|
||||
```
|
||||
|
||||
3. Integrate into agent loop:
|
||||
|
||||
```typescript
|
||||
async run(...) {
|
||||
// Start reasoning timer at beginning
|
||||
this.startReasoningTimer(updateCurrentAiMessage);
|
||||
|
||||
try {
|
||||
// ... agent loop ...
|
||||
|
||||
// When executing a tool:
|
||||
this.addReasoningStep(`Calling ${toolName}...`);
|
||||
updateCurrentAiMessage(this.buildReasoningBlockMarkup());
|
||||
|
||||
// After tool result:
|
||||
this.addReasoningStep(this.summarizeToolResult(toolName, result));
|
||||
updateCurrentAiMessage(this.buildReasoningBlockMarkup());
|
||||
|
||||
// When final response starts:
|
||||
this.stopReasoningTimer();
|
||||
const collapsedBlock = this.buildReasoningBlockMarkup();
|
||||
|
||||
// Stream final response AFTER the collapsed block
|
||||
for await (const chunk of stream) {
|
||||
updateCurrentAiMessage(collapsedBlock + chunk);
|
||||
}
|
||||
|
||||
} finally {
|
||||
this.stopReasoningTimer();
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
4. Add step summarization helper:
|
||||
|
||||
```typescript
|
||||
private summarizeToolResult(toolName: string, result: any): string {
|
||||
switch (toolName) {
|
||||
case 'localSearch':
|
||||
const count = result?.documents?.length || 0;
|
||||
return `Found ${count} relevant note${count !== 1 ? 's' : ''}`;
|
||||
case 'webSearch':
|
||||
return 'Retrieved web search results';
|
||||
case 'getTimeRangeMs':
|
||||
return 'Calculated time range';
|
||||
default:
|
||||
return `Completed ${toolName}`;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Phase 3: React Component
|
||||
|
||||
**File:** `src/components/chat-components/AgentReasoningBlock.tsx` (new)
|
||||
|
||||
```tsx
|
||||
import React, { useState, useEffect, useRef } from "react";
|
||||
|
||||
interface AgentReasoningBlockProps {
|
||||
status: "reasoning" | "collapsed" | "complete";
|
||||
elapsedSeconds: number;
|
||||
steps: string[];
|
||||
isStreaming: boolean;
|
||||
}
|
||||
|
||||
export const AgentReasoningBlock: React.FC<AgentReasoningBlockProps> = ({
|
||||
status,
|
||||
elapsedSeconds,
|
||||
steps,
|
||||
isStreaming,
|
||||
}) => {
|
||||
const [isExpanded, setIsExpanded] = useState(status === "reasoning");
|
||||
|
||||
// Auto-collapse when status changes to collapsed
|
||||
useEffect(() => {
|
||||
if (status === "collapsed" || status === "complete") {
|
||||
setIsExpanded(false);
|
||||
} else if (status === "reasoning") {
|
||||
setIsExpanded(true);
|
||||
}
|
||||
}, [status]);
|
||||
|
||||
const formatTime = (seconds: number) => {
|
||||
if (seconds < 60) return `${seconds}s`;
|
||||
const mins = Math.floor(seconds / 60);
|
||||
const secs = seconds % 60;
|
||||
return `${mins}m ${secs}s`;
|
||||
};
|
||||
|
||||
const isActive = status === "reasoning";
|
||||
|
||||
return (
|
||||
<div className="agent-reasoning-block">
|
||||
{/* Header - always visible */}
|
||||
<div
|
||||
className="agent-reasoning-header"
|
||||
onClick={() => !isActive && setIsExpanded(!isExpanded)}
|
||||
style={{ cursor: isActive ? "default" : "pointer" }}
|
||||
>
|
||||
{/* Spinner or expand chevron */}
|
||||
<span className="agent-reasoning-icon">
|
||||
{isActive ? (
|
||||
<LoadingSpinner />
|
||||
) : (
|
||||
<span className={`chevron ${isExpanded ? "expanded" : ""}`}>▸</span>
|
||||
)}
|
||||
</span>
|
||||
|
||||
{/* Title and timer */}
|
||||
<span className="agent-reasoning-title">{isActive ? "Reasoning" : "Thought for"}</span>
|
||||
<span className="agent-reasoning-timer">{formatTime(elapsedSeconds)}</span>
|
||||
</div>
|
||||
|
||||
{/* Steps - visible when expanded */}
|
||||
{isExpanded && steps.length > 0 && (
|
||||
<ul className="agent-reasoning-steps">
|
||||
{steps.map((step, i) => (
|
||||
<li key={i} className="agent-reasoning-step">
|
||||
{step}
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
const LoadingSpinner: React.FC = () => (
|
||||
<span className="agent-reasoning-spinner">
|
||||
{/* 6-dot braille pattern spinner */}
|
||||
<span className="spinner-dots">⠿</span>
|
||||
</span>
|
||||
);
|
||||
```
|
||||
|
||||
### Phase 4: CSS Styling
|
||||
|
||||
**File:** `src/styles/tailwind.css` (append to existing)
|
||||
|
||||
```css
|
||||
/* Agent Reasoning Block */
|
||||
.agent-reasoning-block {
|
||||
margin: 8px 0;
|
||||
padding: 12px 16px;
|
||||
border-radius: var(--radius-m);
|
||||
background: var(--background-secondary);
|
||||
border: 1px solid var(--background-modifier-border);
|
||||
font-size: var(--font-ui-small);
|
||||
}
|
||||
|
||||
.agent-reasoning-header {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 8px;
|
||||
color: var(--text-muted);
|
||||
}
|
||||
|
||||
.agent-reasoning-icon {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
width: 16px;
|
||||
}
|
||||
|
||||
.agent-reasoning-icon .chevron {
|
||||
transition: transform 0.15s ease;
|
||||
font-size: 10px;
|
||||
}
|
||||
|
||||
.agent-reasoning-icon .chevron.expanded {
|
||||
transform: rotate(90deg);
|
||||
}
|
||||
|
||||
.agent-reasoning-title {
|
||||
font-weight: var(--font-medium);
|
||||
}
|
||||
|
||||
.agent-reasoning-timer {
|
||||
color: var(--text-faint);
|
||||
}
|
||||
|
||||
.agent-reasoning-steps {
|
||||
margin: 8px 0 0 24px;
|
||||
padding: 0;
|
||||
list-style: disc;
|
||||
}
|
||||
|
||||
.agent-reasoning-step {
|
||||
margin: 4px 0;
|
||||
color: var(--text-normal);
|
||||
line-height: 1.4;
|
||||
}
|
||||
|
||||
/* Spinner animation */
|
||||
.agent-reasoning-spinner .spinner-dots {
|
||||
display: inline-block;
|
||||
animation: reasoning-pulse 1s ease-in-out infinite;
|
||||
}
|
||||
|
||||
@keyframes reasoning-pulse {
|
||||
0%,
|
||||
100% {
|
||||
opacity: 0.4;
|
||||
}
|
||||
50% {
|
||||
opacity: 1;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Phase 5: Message Rendering Integration
|
||||
|
||||
**File:** `src/components/chat-components/ChatSingleMessage.tsx`
|
||||
|
||||
**Changes:**
|
||||
|
||||
1. Add parsing for reasoning block marker:
|
||||
|
||||
```typescript
|
||||
function parseAgentReasoningMarker(content: string): {
|
||||
hasReasoning: boolean;
|
||||
status: "reasoning" | "collapsed" | "complete";
|
||||
elapsedSeconds: number;
|
||||
steps: string[];
|
||||
contentAfter: string;
|
||||
} | null {
|
||||
const match = content.match(/<!--AGENT_REASONING:(\w+):(\d+):(.+?)-->/);
|
||||
if (!match) return null;
|
||||
|
||||
const [fullMatch, status, elapsed, stepsJson] = match;
|
||||
const steps = JSON.parse(stepsJson) as string[];
|
||||
|
||||
return {
|
||||
hasReasoning: true,
|
||||
status: status as "reasoning" | "collapsed" | "complete",
|
||||
elapsedSeconds: parseInt(elapsed, 10),
|
||||
steps,
|
||||
contentAfter: content.replace(fullMatch, "").trim(),
|
||||
};
|
||||
}
|
||||
```
|
||||
|
||||
2. Update render logic:
|
||||
|
||||
```typescript
|
||||
// In ChatSingleMessage render:
|
||||
const reasoningData = parseAgentReasoningMarker(message.content);
|
||||
|
||||
return (
|
||||
<div className="chat-message">
|
||||
{/* Agent Reasoning Block (if present) */}
|
||||
{reasoningData?.hasReasoning && (
|
||||
<AgentReasoningBlock
|
||||
status={reasoningData.status}
|
||||
elapsedSeconds={reasoningData.elapsedSeconds}
|
||||
steps={reasoningData.steps}
|
||||
isStreaming={isStreaming}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Message content (after reasoning marker) */}
|
||||
<div className="message-content">
|
||||
{/* Render remainingContent via markdown */}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
```
|
||||
|
||||
### Phase 6: Remove Old Tool Call Banner
|
||||
|
||||
**Files to modify:**
|
||||
|
||||
1. `src/components/chat-components/ToolCallBanner.tsx` - Delete or deprecate
|
||||
2. `src/components/chat-components/toolCallRootManager.tsx` - Simplify or remove
|
||||
3. `src/LLMProviders/chainRunner/utils/toolCallParser.ts` - Keep for backward compat with saved messages
|
||||
4. `src/LLMProviders/chainRunner/utils/ThinkBlockStreamer.ts` - Remove tool call marker handling
|
||||
|
||||
**Deprecation strategy:**
|
||||
|
||||
- Keep `parseToolCallMarkers()` for rendering old saved messages
|
||||
- Remove `createToolCallMarker()` and `updateToolCallMarker()`
|
||||
- Don't create new tool call markers in agent runner
|
||||
|
||||
### Phase 7: Disable Thinking Block in Agent Mode
|
||||
|
||||
**File:** `src/LLMProviders/chainRunner/utils/ThinkBlockStreamer.ts`
|
||||
|
||||
Add flag to skip thinking content extraction in agent mode:
|
||||
|
||||
```typescript
|
||||
export class ThinkBlockStreamer {
|
||||
private suppressThinkingContent: boolean;
|
||||
|
||||
constructor(
|
||||
updateFn: (message: string) => void,
|
||||
options?: { suppressThinkingContent?: boolean }
|
||||
) {
|
||||
this.updateFn = updateFn;
|
||||
this.suppressThinkingContent = options?.suppressThinkingContent ?? false;
|
||||
}
|
||||
|
||||
processChunk(chunk: any): void {
|
||||
if (this.suppressThinkingContent) {
|
||||
// Skip thinking content, only extract text
|
||||
// ... simplified logic ...
|
||||
} else {
|
||||
// ... existing thinking block logic ...
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
In AutonomousAgentChainRunner:
|
||||
|
||||
```typescript
|
||||
const streamer = new ThinkBlockStreamer(updateCurrentAiMessage, {
|
||||
suppressThinkingContent: true, // Agent mode uses AgentReasoningBlock instead
|
||||
});
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Performance Considerations
|
||||
|
||||
1. **Timer Updates**: Use 100ms interval for responsive feel without excessive re-renders
|
||||
2. **Step Limit**: Keep only last 2 steps to prevent DOM bloat
|
||||
3. **Marker Format**: Compact JSON for minimal message overhead
|
||||
4. **React Roots**: Use same pattern as tool call banner for persistent roots
|
||||
|
||||
---
|
||||
|
||||
## Migration Path
|
||||
|
||||
1. **Phase 1**: Add new AgentReasoningBlock alongside existing tool call banner
|
||||
2. **Phase 2**: Test with agent mode, ensure backward compat with old messages
|
||||
3. **Phase 3**: Remove tool call banner creation code (keep parsing)
|
||||
4. **Phase 4**: Clean up deprecated code after stable release
|
||||
|
||||
---
|
||||
|
||||
## Testing Checklist
|
||||
|
||||
- [ ] Timer updates smoothly (no flicker)
|
||||
- [ ] Steps appear as tools execute
|
||||
- [ ] Block collapses when final response starts
|
||||
- [ ] Collapsed block shows correct elapsed time
|
||||
- [ ] Click to expand works after response complete
|
||||
- [ ] Old messages with tool call banners still render
|
||||
- [ ] No thinking blocks appear in agent mode
|
||||
- [ ] Performance: no lag with multiple iterations
|
||||
|
||||
---
|
||||
|
||||
## File Summary
|
||||
|
||||
| File | Action |
|
||||
| ------------------------------- | ------------------------------------- |
|
||||
| `AgentReasoningState.ts` | **New** - State management |
|
||||
| `AgentReasoningBlock.tsx` | **New** - React component |
|
||||
| `AutonomousAgentChainRunner.ts` | **Modify** - Add reasoning tracking |
|
||||
| `ChatSingleMessage.tsx` | **Modify** - Render reasoning block |
|
||||
| `ThinkBlockStreamer.ts` | **Modify** - Add suppress option |
|
||||
| `tailwind.css` | **Modify** - Add styles |
|
||||
| `ToolCallBanner.tsx` | **Deprecate** - Keep for old messages |
|
||||
| `toolCallParser.ts` | **Keep** - Backward compat only |
|
||||
51
designdocs/todo/TECHDEBT.md
Normal file
|
|
@ -0,0 +1,51 @@
|
|||
# TODO - Technical Debt & Future Improvements
|
||||
|
||||
This document tracks technical debt items and improvements that need to be addressed in the future.
|
||||
|
||||
## 1. Docs4LLM SSL Error in Projects Mode
|
||||
|
||||
### Issue Description
|
||||
|
||||
Document parsing for projects mode is failing with an SSL error (`net::ERR_SSL_BAD_RECORD_MAC_ALERT`) when trying to upload files to the docs4llm API endpoint.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **Error Location**: `src/LLMProviders/brevilabsClient.ts:185` in `makeFormDataRequest` method
|
||||
- **Root Cause**: The method uses native `fetch` API instead of Obsidian's `requestUrl` API
|
||||
- **Context**: While regular JSON requests were migrated to use `safeFetch` (which uses `requestUrl`) in commit e49aafa to fix CORS issues, the `makeFormDataRequest` method was not updated
|
||||
|
||||
### Why Current Approaches Won't Work
|
||||
|
||||
1. **Backend Constraint**: The `/docs4llm` endpoint only accepts multipart/form-data format with `files: List[UploadFile]`
|
||||
2. **Obsidian Limitation**: The existing `safeFetch` function is hardcoded for `application/json` content type
|
||||
3. **No JSON Alternative**: Unlike `pdf4llm` which accepts base64 JSON, there's no JSON endpoint for `docs4llm`
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Create a new `safeFetchFormData` function that:
|
||||
|
||||
1. Uses Obsidian's `requestUrl` API with proper multipart/form-data configuration
|
||||
2. Handles FormData objects correctly
|
||||
3. Bypasses CORS and SSL restrictions like `safeFetch` does for JSON
|
||||
|
||||
### Alternative Solutions
|
||||
|
||||
1. **Backend Modification**: Add a new `/docs4llm-base64` endpoint that accepts base64-encoded JSON payloads
|
||||
2. **Research Obsidian API**: Investigate if newer versions of Obsidian's `requestUrl` support multipart/form-data
|
||||
3. **SSL Certificate Fix**: Address the underlying SSL certificate issue (temporary workaround)
|
||||
|
||||
### Impact
|
||||
|
||||
- Users cannot parse non-markdown files (PDFs, Word docs, etc.) in projects mode
|
||||
- This affects the core functionality of project context loading
|
||||
- Workaround: Users must ensure their projects only contain markdown files
|
||||
|
||||
### References
|
||||
|
||||
- Related commit: e49aafa (Brevilabs CORS issue #918)
|
||||
- Forum discussion: https://forum.obsidian.md/t/holo-how-to-add-a-png-image-or-file-to-formdata-in-obsidian-like-below-this-help/73420
|
||||
- Backend implementation: `/Users/chaoyang/webapps/brevilabs-api/app/main.py:1039`
|
||||
|
||||
---
|
||||
|
||||
_Last updated: 2025-07-18_
|
||||
152
designdocs/todo/TODO-composer-tool-redesign.md
Normal file
|
|
@ -0,0 +1,152 @@
|
|||
# TODO: Composer Tool Redesign for Faster Feedback
|
||||
|
||||
**Date:** 2026-02-04
|
||||
**Assignee:** @wenzhengjiang
|
||||
**Status:** TODO
|
||||
|
||||
## Problem
|
||||
|
||||
The `writeToFile` and `replaceInFile` composer tools have a significant UX issue: there's a **30+ second gap** between the initial reasoning step and any meaningful feedback about what the agent intends to do.
|
||||
|
||||
### Root Cause
|
||||
|
||||
The current flow requires the model to generate the **entire modified file content** in a single tool call:
|
||||
|
||||
```
|
||||
User: "Remove all the headings"
|
||||
↓
|
||||
Model receives prompt + full file content
|
||||
↓
|
||||
Model generates ENTIRE modified file (30+ seconds)
|
||||
↓
|
||||
Tool call emitted with full content
|
||||
↓
|
||||
UI finally shows "Writing to filename..."
|
||||
```
|
||||
|
||||
During those 30+ seconds, the user has no idea what the agent is planning to do.
|
||||
|
||||
### Current Tool Schema
|
||||
|
||||
```typescript
|
||||
writeToFile({
|
||||
path: string,
|
||||
content: string | object, // ← ENTIRE file content generated upfront
|
||||
confirmation: boolean,
|
||||
});
|
||||
```
|
||||
|
||||
## Proposed Solution: Multi-Step Tool Design
|
||||
|
||||
Split the composer operation into two phases:
|
||||
|
||||
### Phase 1: Intent Declaration (Fast)
|
||||
|
||||
A lightweight tool call that declares intent without generating content:
|
||||
|
||||
```typescript
|
||||
declareEditIntent({
|
||||
path: string,
|
||||
operation: "rewrite" | "modify" | "create",
|
||||
description: string, // e.g., "Remove all headings from the document"
|
||||
});
|
||||
```
|
||||
|
||||
This would return almost immediately (1-2 seconds) because the model only needs to decide WHAT to do, not generate the full content.
|
||||
|
||||
**UI shows:** "Planning to remove all headings from Daily-AI-Digest.md..."
|
||||
|
||||
### Phase 2: Content Generation (Slow but expected)
|
||||
|
||||
After intent is confirmed, the full content generation happens:
|
||||
|
||||
```typescript
|
||||
executeEdit({
|
||||
path: string,
|
||||
content: string | object,
|
||||
});
|
||||
```
|
||||
|
||||
**UI shows:** "Generating changes..." → "Writing to filename..."
|
||||
|
||||
### Benefits
|
||||
|
||||
1. **Immediate feedback**: User knows the intent within 1-2 seconds
|
||||
2. **Opportunity to cancel**: User can abort before expensive generation
|
||||
3. **Better UX**: Progress feels natural (planning → executing)
|
||||
4. **Clearer reasoning steps**: Each phase has distinct, meaningful steps
|
||||
|
||||
## Additional Requirements
|
||||
|
||||
### Diff History Cache for Reliable Revert
|
||||
|
||||
Cache the last N diffs per file to enable reliable undo/revert functionality:
|
||||
|
||||
- Store diffs (not full file snapshots) to minimize storage
|
||||
- Keep last N changes (e.g., N=10) per file path
|
||||
- Enable "Revert last change" and "Revert to version X" actions
|
||||
- Clear old diffs when limit exceeded (FIFO)
|
||||
|
||||
```typescript
|
||||
interface DiffHistoryEntry {
|
||||
timestamp: number;
|
||||
path: string;
|
||||
diff: Change[]; // from 'diff' library
|
||||
description: string; // e.g., "Remove all headings"
|
||||
}
|
||||
```
|
||||
|
||||
### Dedicated "Apply" Model
|
||||
|
||||
Consider using a separate small, fast, and cheap model specifically for applying diffs:
|
||||
|
||||
- **Main model**: Generates the intent and edit description (Phase 1)
|
||||
- **Apply model**: Executes the actual file transformation (Phase 2)
|
||||
|
||||
Benefits:
|
||||
|
||||
- Faster execution for Phase 2 (smaller model = faster inference)
|
||||
- Lower cost (cheap model for mechanical transformation)
|
||||
- Main model focuses on understanding, apply model focuses on execution
|
||||
- Could use a fine-tuned model optimized for code/text transformations
|
||||
|
||||
Candidates: Gemini Flash Lite or comparable models.
|
||||
|
||||
### Structured Tool Results
|
||||
|
||||
Currently, ComposerTools returns plain strings for errors/no-ops, making result parsing fragile:
|
||||
|
||||
```typescript
|
||||
// Current: Plain strings (fragile)
|
||||
return `File is too small to use this tool...`;
|
||||
return `Search text not found in file ${path}...`;
|
||||
return `No changes made to ${path}...`;
|
||||
```
|
||||
|
||||
The reasoning UI (`AgentReasoningState.ts`) uses string matching to detect these cases, which breaks if messages change.
|
||||
|
||||
**Proposed:** Return structured results:
|
||||
|
||||
```typescript
|
||||
interface ComposerToolResult {
|
||||
status: "success" | "rejected" | "no-op" | "error";
|
||||
message: string;
|
||||
path?: string;
|
||||
}
|
||||
```
|
||||
|
||||
This enables reliable status detection in the reasoning UI without fragile string matching.
|
||||
|
||||
## Related Files
|
||||
|
||||
- `src/tools/ComposerTools.ts` - Current tool implementations
|
||||
- `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts` - ReAct loop
|
||||
- `src/LLMProviders/chainRunner/utils/AgentReasoningState.ts` - Reasoning UI state
|
||||
|
||||
## Next Steps
|
||||
|
||||
1. Design the multi-step tool API
|
||||
2. Prototype with `writeToFile` first
|
||||
3. Update reasoning UI to handle two-phase operations
|
||||
4. Test with various file sizes and operations
|
||||
5. Roll out to `replaceInFile`
|
||||
311
designdocs/todo/TOKEN_BUDGET_ENFORCEMENT.md
Normal file
|
|
@ -0,0 +1,311 @@
|
|||
# Token Budget Enforcement
|
||||
|
||||
## Table of Contents
|
||||
|
||||
1. [Problem Statement](#problem-statement)
|
||||
2. [Current Compaction Architecture](#current-compaction-architecture)
|
||||
3. [Root Cause Analysis](#root-cause-analysis)
|
||||
4. [Fix Plan](#fix-plan)
|
||||
5. [References](#references)
|
||||
|
||||
---
|
||||
|
||||
## Problem Statement
|
||||
|
||||
The model proxy receives requests with token counts far exceeding the model's context window (e.g., 2.7M tokens sent to a 1M-token Vertex AI model). The plugin's auto-compaction system was expected to prevent this but fails because **no compaction mechanism checks the total assembled payload** — each compactor guards only its own subset.
|
||||
|
||||
```
|
||||
ContextWindowExceededError: The input token count (2769478)
|
||||
exceeds the maximum number of tokens allowed (1048575).
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Current Compaction Architecture
|
||||
|
||||
There are **three separate compaction mechanisms** in the plugin. None of them enforce a total token budget against the `autoCompactThreshold` setting.
|
||||
|
||||
### 1. Turn-Time Context Compaction (ContextCompactor)
|
||||
|
||||
**Where**: `ContextManager.processMessageContext()` (`src/core/ContextManager.ts:231-258`)
|
||||
**When**: Every time a user message is processed, before the envelope is built.
|
||||
**What it covers**: L2 (previous turn context) + L3 (current turn context) combined.
|
||||
**What it does NOT cover**: L1 (system prompt), L4 (chat history), L5 (user message).
|
||||
|
||||
```
|
||||
Trigger condition:
|
||||
(processedUserMessage + contextPortion).length > autoCompactThreshold * 4
|
||||
|
||||
Where:
|
||||
autoCompactThreshold = settings.autoCompactThreshold (default: 128,000 tokens)
|
||||
charThreshold = 128,000 * 4 = 512,000 chars
|
||||
```
|
||||
|
||||
When triggered, `ContextCompactor.compact()` performs map-reduce LLM summarization on individual XML blocks larger than 50k chars. The user message itself is never compacted.
|
||||
|
||||
**Key limitation**: This threshold check measures `processedUserMessage + contextPortion` (which is L5 + L2 + L3). It does NOT include:
|
||||
|
||||
- L1 (system prompt) — typically 2-10k tokens
|
||||
- L4 (chat history) — potentially **hundreds of thousands of tokens**
|
||||
|
||||
### 2. L2 Carry-Forward Compaction (L2ContextCompactor)
|
||||
|
||||
**Where**: `ContextManager.compactSegmentForL2()` (`src/core/ContextManager.ts:706-733`)
|
||||
**When**: When previous turn L3 segments are promoted into L2 for the next turn.
|
||||
**What it does**: Deterministic structure+preview compression (headings + truncated sections). No LLM calls.
|
||||
|
||||
This is a **per-segment** operation that reduces each context artifact to a `<prior_context>` block with ~500 chars per section. This prevents L2 from growing unbounded as turns accumulate.
|
||||
|
||||
### 3. Chat History Compaction (ChatHistoryCompactor)
|
||||
|
||||
**Where**: `MemoryManager.saveContext()` (`src/LLMProviders/memoryManager.ts:61-72`)
|
||||
**When**: After each assistant response, at memory save time.
|
||||
**What it does**: Compacts tool results (`localSearch`, `readNote`, etc.) in assistant responses before saving to `BufferWindowMemory`.
|
||||
|
||||
This compacts **only the tool-result portions** of assistant messages. The rest of the assistant text and all user messages are stored verbatim.
|
||||
|
||||
### Summary: What Each System Protects
|
||||
|
||||
| Compaction System | Scope | Token-Aware? | Covers Full Payload? |
|
||||
| ---------------------------------- | ---------------------------------- | ------------------------------- | ---------------------- |
|
||||
| ContextCompactor (turn-time) | L2 + L3 context XML blocks | Threshold-based (char estimate) | No — misses L1, L4, L5 |
|
||||
| L2ContextCompactor (carry-forward) | Individual L2 segments | No — fixed per-segment | No — per-segment only |
|
||||
| ChatHistoryCompactor (save-time) | Tool results in assistant messages | No — fixed size | No — only tool results |
|
||||
|
||||
---
|
||||
|
||||
## Root Cause Analysis
|
||||
|
||||
### The Core Problem: No Total Payload Budget
|
||||
|
||||
The critical gap is **systemic**: no compaction mechanism checks the total assembled payload (L1+L2+L3+L4+L5) against any budget. Each compactor guards only its own subset, and no final safety net exists.
|
||||
|
||||
### What L4 Actually Contains
|
||||
|
||||
L4 (chat history) is often assumed to be the main token consumer, but investigation shows it is relatively well-controlled:
|
||||
|
||||
- **User messages in L4** = bare L5 text only (no context XML). `BaseChainRunner.handleResponse()` extracts `l5Text` from the envelope and saves only that to memory.
|
||||
- **Assistant responses in L4** = compacted at save time by `ChatHistoryCompactor`, which strips tool result XML (`localSearch`, `readNote`, `note_context`, etc.).
|
||||
- **Agent-mode responses**: `AutonomousAgentChainRunner` saves only `loopResult.finalResponse` (the final answer), NOT the full reasoning/tool-call chain.
|
||||
|
||||
L4 does grow with conversation length, but it is not unbounded — `BufferWindowMemory` limits it to `k = contextTurns * 2` messages (default: 30), and both user and assistant sides are relatively compact.
|
||||
|
||||
### The Real Culprits: L1 and Unchecked Layer Accumulation
|
||||
|
||||
The overflow happens because **multiple layers accumulate without any shared budget**:
|
||||
|
||||
#### L1: Project Context Is Never Budgeted
|
||||
|
||||
In Projects mode, `ChatManager.getSystemPromptForMessage()` concatenates all project files, web content, and YouTube transcripts into a `<project_context>` block inside L1. This can easily reach **hundreds of thousands of tokens** for large projects.
|
||||
|
||||
L1 is **never compacted by any system** — no compactor even sees it.
|
||||
|
||||
#### Compaction Threshold Is Blind to L1
|
||||
|
||||
`ContextManager.processMessageContext()` uses a hardcoded `PROJECT_COMPACT_THRESHOLD = 1,000,000` tokens for Projects mode compaction. This threshold checks only L2+L3 size — it is completely blind to L1 (project context) size. It is set as if L2+L3 is the _entire_ budget, when in reality L1 may have already consumed most of the available context window.
|
||||
|
||||
For non-project chains, `autoCompactThreshold` (default 128k) is used, but it also only checks L2+L3.
|
||||
|
||||
#### L4: No Budget Awareness
|
||||
|
||||
`loadAndAddChatHistory()` loads all history messages without checking how much token budget remains after L1+L2+L3+L5 are assembled:
|
||||
|
||||
```typescript
|
||||
export async function loadAndAddChatHistory(
|
||||
memory: any,
|
||||
messages: Array<{ role: string; content: any }>
|
||||
): Promise<ProcessedMessage[]> {
|
||||
const memoryVariables = await memory.loadMemoryVariables({});
|
||||
const rawHistory = memoryVariables.history || [];
|
||||
// ... processes and adds ALL history messages with NO size check
|
||||
}
|
||||
```
|
||||
|
||||
### How 2.7M Tokens Happen
|
||||
|
||||
In a Projects-mode conversation:
|
||||
|
||||
```
|
||||
L1 (system + project_context): ~500k tokens ← UNBUDGETED, never compacted
|
||||
L2 (previous context, compacted): ~20k tokens
|
||||
L3 (current turn context): ~50k tokens
|
||||
─────────
|
||||
ContextCompactor checks L2+L3: 70k < 1,000k threshold → NO compaction triggered
|
||||
(threshold is blind to 500k in L1)
|
||||
|
||||
L4 (15 turns of chat history): ~200k tokens ← loaded with no remaining budget check
|
||||
L5 (user message): ~2k tokens
|
||||
─────────────────────────────────────────────────
|
||||
TOTAL: ~772k tokens → may exceed model's context window
|
||||
```
|
||||
|
||||
In extreme cases (large projects + long conversations + heavy context attachments), totals can reach 2M+ tokens.
|
||||
|
||||
### All Chain Runners Are Affected
|
||||
|
||||
All chain runners call `loadAndAddChatHistory()` without any token budget:
|
||||
|
||||
| Runner | File | Line |
|
||||
| -------------------------- | ------------------------------------------------------------ | ---- |
|
||||
| LLMChainRunner | `src/LLMProviders/chainRunner/LLMChainRunner.ts` | 45 |
|
||||
| CopilotPlusChainRunner | `src/LLMProviders/chainRunner/CopilotPlusChainRunner.ts` | 606 |
|
||||
| AutonomousAgentChainRunner | `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts` | 597 |
|
||||
| VaultQAChainRunner | `src/LLMProviders/chainRunner/VaultQAChainRunner.ts` | 191 |
|
||||
|
||||
### The `contextTurns` Setting Is a Poor Proxy
|
||||
|
||||
`BufferWindowMemory` is configured with `k = contextTurns * 2` (default: 30 messages). This is a crude count-based limit that:
|
||||
|
||||
- Has no relation to actual token consumption
|
||||
- Cannot adapt to varying message sizes
|
||||
- Provides no guarantees about total payload size
|
||||
|
||||
A token-based budget for L4 makes `contextTurns` redundant.
|
||||
|
||||
---
|
||||
|
||||
## Fix Plan
|
||||
|
||||
### Guiding Principles
|
||||
|
||||
1. **Model-agnostic**: The plugin supports many LLM providers. No model-specific context window logic. Use `autoCompactThreshold` (user-configurable) as the single total budget.
|
||||
2. **Single enforcement point**: Token budget must be checked where all layers are assembled, not scattered across individual compactors.
|
||||
3. **History guarantee**: The LLM must always see at least some recent chat history to resume conversation context, even when L1+L2+L3 consume most of the budget.
|
||||
4. **Graceful degradation**: When over budget, drop the least-valuable content first (oldest history turns), then compact further if needed.
|
||||
5. **Backwards compatible**: Existing compaction systems remain; this adds a final safety net.
|
||||
6. **No LLM calls in the hot path**: Budget enforcement should use fast char-based estimation (chars / 4), not LLM summarization.
|
||||
|
||||
### Phase 1: Token Budget Guard (Critical Fix)
|
||||
|
||||
**Goal**: Prevent over-budget payloads from ever reaching the LLM.
|
||||
|
||||
#### 1.1 Make ContextManager L1-Aware
|
||||
|
||||
Currently `ContextManager.processMessageContext()` checks `(L2+L3).length > threshold * 4` where threshold is either `autoCompactThreshold` or `PROJECT_COMPACT_THRESHOLD`. Both are blind to L1 size.
|
||||
|
||||
**Fix**: The compaction threshold for L2+L3 must account for L1:
|
||||
|
||||
```
|
||||
effectiveThreshold = autoCompactThreshold - estimateTokens(L1)
|
||||
```
|
||||
|
||||
This ensures that when L1 is large (e.g., Projects mode with many files), L2+L3 compaction triggers earlier, leaving room for L4 and L5.
|
||||
|
||||
**Kill `PROJECT_COMPACT_THRESHOLD`** — it is a hardcoded 1M value that pretends L1 doesn't exist. Replace with the same `autoCompactThreshold - L1` formula for all chain types.
|
||||
|
||||
**File**: `src/core/ContextManager.ts`
|
||||
|
||||
#### 1.2 Add Token Budget to `loadAndAddChatHistory()`
|
||||
|
||||
Add an optional `tokenBudget` parameter to `loadAndAddChatHistory()`. When provided:
|
||||
|
||||
1. Load all history messages from `BufferWindowMemory`
|
||||
2. Estimate token count of each message (chars / 4)
|
||||
3. Drop oldest complete turns (user+assistant pairs) until cumulative total fits within budget
|
||||
4. Always keep at least the most recent turn (history guarantee)
|
||||
5. Log a warning when turns are dropped
|
||||
|
||||
```
|
||||
Token Budget Allocation:
|
||||
autoCompactThreshold (e.g., 128,000 tokens)
|
||||
- estimateTokens(L1) system prompt + project context
|
||||
- estimateTokens(L2) previous context library
|
||||
- estimateTokens(L3) current turn context
|
||||
- estimateTokens(L5) user message
|
||||
- reservedForOutput (~4,096 for response generation)
|
||||
= remaining budget for L4 chat history
|
||||
```
|
||||
|
||||
**File**: `src/LLMProviders/chainRunner/utils/chatHistoryUtils.ts`
|
||||
|
||||
#### 1.3 Update All Chain Runners
|
||||
|
||||
Each chain runner calls `loadAndAddChatHistory()`. Update call sites to:
|
||||
|
||||
1. Calculate the token size of already-assembled non-L4 messages (L1+L2+L3+L5)
|
||||
2. Compute `historyBudget = autoCompactThreshold - nonL4Tokens - outputReserve`
|
||||
3. Pass `historyBudget` to `loadAndAddChatHistory()`
|
||||
|
||||
**Files**:
|
||||
|
||||
- `src/LLMProviders/chainRunner/LLMChainRunner.ts`
|
||||
- `src/LLMProviders/chainRunner/CopilotPlusChainRunner.ts`
|
||||
- `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts`
|
||||
- `src/LLMProviders/chainRunner/VaultQAChainRunner.ts`
|
||||
|
||||
#### 1.4 Deprecate `contextTurns` Setting
|
||||
|
||||
Token-based budget trimming makes the count-based `contextTurns` setting redundant.
|
||||
|
||||
- Replace `BufferWindowMemory.k = contextTurns * 2` with a generous internal constant (e.g., `k = 100`)
|
||||
- The token budget in step 1.2 handles the actual trimming
|
||||
- Remove the "Conversation turns in context" slider from `ModelSettings.tsx`
|
||||
|
||||
**Files**:
|
||||
|
||||
- `src/LLMProviders/memoryManager.ts`
|
||||
- `src/settings/v2/components/ModelSettings.tsx`
|
||||
|
||||
### Phase 2: Smarter History Trimming (Enhancement)
|
||||
|
||||
**Goal**: When budget is tight, trim intelligently rather than just dropping oldest turns.
|
||||
|
||||
#### 2.1 Prioritized trimming strategy
|
||||
|
||||
When over budget, apply in order:
|
||||
|
||||
1. **Drop oldest complete turns** (user+assistant pairs) from L4
|
||||
2. **Truncate remaining long assistant responses** in L4 (keep first N chars)
|
||||
3. If _still_ over budget after L4 is minimized, **warn user** and proceed — the LLM will still see the most recent turn
|
||||
|
||||
### Phase 3: Observability (Enhancement)
|
||||
|
||||
#### 3.1 Surface token usage to UI
|
||||
|
||||
Add a debug/info display showing:
|
||||
|
||||
- Estimated tokens per layer (L1, L2, L3, L4, L5)
|
||||
- Total vs. `autoCompactThreshold`
|
||||
- Whether any history turns were dropped
|
||||
|
||||
This helps users understand why responses might miss context from earlier turns.
|
||||
|
||||
### Implementation Order
|
||||
|
||||
| Step | Description | Files Changed | Risk |
|
||||
| ---- | ----------------------------------------- | ------------------------------- | ------ |
|
||||
| 1.1 | L1-aware compaction threshold | ContextManager.ts | Medium |
|
||||
| 1.2 | Token budget in `loadAndAddChatHistory()` | chatHistoryUtils.ts | Medium |
|
||||
| 1.3 | Update chain runner call sites | 4 chain runner files | Medium |
|
||||
| 1.4 | Deprecate `contextTurns` | memoryManager.ts, ModelSettings | Low |
|
||||
| 2.1 | Prioritized trimming | chatHistoryUtils.ts | Low |
|
||||
| 3.1 | Token usage debug display | UI components | Low |
|
||||
|
||||
Phase 1 (steps 1.1-1.4) is the **critical fix** that prevents the overflow. Phases 2-3 are improvements.
|
||||
|
||||
---
|
||||
|
||||
## References
|
||||
|
||||
### Source Files
|
||||
|
||||
| File | Role |
|
||||
| ------------------------------------------------------------ | ----------------------------------------------- |
|
||||
| `src/core/ContextManager.ts` | Turn-time compaction trigger (L2+L3) |
|
||||
| `src/core/ContextCompactor.ts` | Map-reduce LLM summarization |
|
||||
| `src/context/L2ContextCompactor.ts` | Deterministic L2 segment compaction |
|
||||
| `src/context/ChatHistoryCompactor.ts` | Tool result compaction at save time |
|
||||
| `src/LLMProviders/memoryManager.ts` | Memory save with compaction |
|
||||
| `src/LLMProviders/chainRunner/utils/chatHistoryUtils.ts` | Chat history loading (no budget) |
|
||||
| `src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts` | Agent message assembly |
|
||||
| `src/LLMProviders/chainRunner/CopilotPlusChainRunner.ts` | CopilotPlus message assembly |
|
||||
| `src/LLMProviders/chainRunner/LLMChainRunner.ts` | Basic LLM message assembly |
|
||||
| `src/LLMProviders/chainRunner/VaultQAChainRunner.ts` | VaultQA message assembly |
|
||||
| `src/LLMProviders/chatModelManager.ts` | Chat model management |
|
||||
| `src/constants.ts` | Default settings (autoCompactThreshold: 128000) |
|
||||
|
||||
### Related Docs
|
||||
|
||||
- [CONTEXT_ENGINEERING.md](./CONTEXT_ENGINEERING.md) — L1-L5 layer architecture
|
||||
- [MESSAGE_ARCHITECTURE.md](./MESSAGE_ARCHITECTURE.md) — Message flow and storage
|
||||
- [TECHDEBT.md](./TECHDEBT.md) — Known technical debt
|
||||
446
designdocs/todo/UI_RENDERING_PERFORMANCE.md
Normal file
|
|
@ -0,0 +1,446 @@
|
|||
# TODO - UI Rendering Performance Issues
|
||||
|
||||
This document tracks UI rendering performance issues identified through a comprehensive audit of the React component tree, state management, and streaming paths. Findings are ranked by severity and organized by recommended fix priority.
|
||||
|
||||
## Severity Legend
|
||||
|
||||
- **CRITICAL** - Causes visible jank/stalls during normal usage; affects every user
|
||||
- **HIGH** - Causes noticeable stalls in specific scenarios or degrades with scale
|
||||
- **MEDIUM** - Contributes to cumulative performance degradation
|
||||
- **LOW** - Minor inefficiency; fix opportunistically
|
||||
|
||||
---
|
||||
|
||||
## 1. [TODO] ChatSingleMessage Not Memoized — Streaming Re-renders All Historical Messages (CRITICAL)
|
||||
|
||||
### Issue Description
|
||||
|
||||
`ChatSingleMessage` is the most expensive component in the application yet is not wrapped in `React.memo`. Every streaming token update causes ALL historical messages to re-render with full MarkdownRenderer passes and DOM manipulation.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **Files**: `src/components/chat-components/ChatSingleMessage.tsx`, `src/components/chat-components/ChatMessages.tsx:88-115`
|
||||
- **Root Cause**: When `ChatMessages` (which IS memoized) re-renders due to `currentAiMessage` changing during streaming, the `.map()` at line 88 creates new React elements for ALL historical messages. Each gets new inline closure props:
|
||||
- `() => onRegenerate(index)` (line 108)
|
||||
- `(newMessage) => onEdit(index, newMessage)` (line 109)
|
||||
- `() => onDelete(index)` (line 110)
|
||||
- These inline closures create new function references on every render, which would defeat `React.memo` even if it were added without also stabilizing the callbacks.
|
||||
- `ChatSingleMessage` contains: `MarkdownRenderer.renderMarkdown()`, DOM manipulation (`querySelectorAll`, `createElement`, `insertBefore`), `parseToolCallMarkers()`, multiple regex passes, and multiple `useEffect` hooks.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
1. Wrap `ChatSingleMessage` in `React.memo` with a custom comparator that checks `message.id`, `message.message`, `isStreaming`, and callback identity.
|
||||
2. Replace inline closure callbacks in `ChatMessages.map()` with stable references. Options:
|
||||
- Pass `messageIndex` as a prop and let `ChatSingleMessage` call `onRegenerate(messageIndex)` internally.
|
||||
- Use `useCallback` with a ref-based pattern to avoid `chatHistory` dependency (see Finding #9).
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Every LLM response stream for every user.
|
||||
- **Severity scales with**: Conversation length. A 20-message conversation means 20 unnecessary expensive re-renders per animation frame during streaming.
|
||||
- **Expected improvement**: Eliminating historical message re-renders during streaming would be the single largest performance win in the codebase.
|
||||
|
||||
---
|
||||
|
||||
## 2. [TODO] O(N^2) Filter Inside .map() in ChatMessages (CRITICAL)
|
||||
|
||||
### Issue Description
|
||||
|
||||
`chatHistory.filter()` is called inside the `.map()` callback on every iteration, creating O(N^2) complexity per render.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/chat-components/ChatMessages.tsx:89`
|
||||
- **Code**: `const visibleMessages = chatHistory.filter((m) => m.isVisible);` is called inside `.map()` to compute `isLastMessage`. For N messages, this executes N filter operations of O(N) each = O(N^2).
|
||||
- Combined with Finding #1 (re-renders on every streaming frame), this compounds badly.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Hoist the filter before the `.map()`:
|
||||
|
||||
```tsx
|
||||
const visibleMessages = useMemo(() => chatHistory.filter((m) => m.isVisible), [chatHistory]);
|
||||
// Then use visibleMessages.length inside .map()
|
||||
```
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Every render during streaming.
|
||||
- **For 100 messages**: 10,000 filter operations per render frame.
|
||||
- **Fix effort**: Trivial — single-line hoist.
|
||||
|
||||
---
|
||||
|
||||
## 3. [TODO] ChatManager.getCurrentMessageRepo() Creates New ChatPersistenceManager on Every Call (HIGH)
|
||||
|
||||
### Issue Description
|
||||
|
||||
A new `ChatPersistenceManager` object is allocated on every call to `getCurrentMessageRepo()`, which is invoked by virtually every read/write operation.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/core/ChatManager.ts:82-87`
|
||||
- **Code**: `this.persistenceManager = new ChatPersistenceManager(this.plugin.app, currentRepo, this.chainManager)` runs unconditionally inside `getCurrentMessageRepo()`.
|
||||
- This method is called by `getDisplayMessages()`, `getLLMMessages()`, `getMessage()`, `addMessage()`, `deleteMessage()`, etc.
|
||||
- Via `useChatManager`, `getDisplayMessages()` is called on every subscription notification from `ChatUIState`.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Cache the `ChatPersistenceManager` per project key. Only recreate when the project actually changes:
|
||||
|
||||
```typescript
|
||||
if (!this.persistenceManagers.has(projectKey)) {
|
||||
this.persistenceManagers.set(projectKey, new ChatPersistenceManager(...));
|
||||
}
|
||||
```
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Every message operation (read or write).
|
||||
- **Causes**: Unnecessary object allocation and GC pressure on every render cycle.
|
||||
|
||||
---
|
||||
|
||||
## 4. [TODO] useChatManager Creates New Array Reference on Every State Notification (HIGH)
|
||||
|
||||
### Issue Description
|
||||
|
||||
`useChatManager` always spreads into a new array, meaning React always sees a new `messages` reference, defeating downstream memoization.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/hooks/useChatManager.ts:21`
|
||||
- **Code**: `setMessages([...chatUIState.getMessages()])` — the spread operator always creates a new array reference regardless of whether the content has changed.
|
||||
- Every `notifyListeners()` call (from any message operation) triggers this, causing `ChatMessages` to re-render even if the actual messages haven't changed.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Use structural comparison or a version counter to avoid unnecessary state updates:
|
||||
|
||||
```typescript
|
||||
const unsubscribe = chatUIState.subscribe(() => {
|
||||
const next = chatUIState.getMessages();
|
||||
setMessages((prev) => {
|
||||
// Only update if messages actually changed
|
||||
if (
|
||||
prev.length === next.length &&
|
||||
prev.every((m, i) => m.id === next[i].id && m.message === next[i].message)
|
||||
) {
|
||||
return prev;
|
||||
}
|
||||
return [...next];
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
Alternatively, add a version/generation counter to `MessageRepository` and only spread when the version changes.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Every state change cascades into unnecessary `ChatMessages` re-renders.
|
||||
- **Compounds with**: Finding #1 (unmemoized ChatSingleMessage) and Finding #9 (unstable callback deps).
|
||||
|
||||
---
|
||||
|
||||
## 5. [TODO] useChatScrolling Triggers Expensive DOM Queries on Every chatHistory Change (HIGH)
|
||||
|
||||
### Issue Description
|
||||
|
||||
`calculateDynamicMinHeight` performs DOM queries (`querySelector`, `getBoundingClientRect`) and is called on every `chatHistory` change, causing layout thrashing during streaming.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/hooks/useChatScrolling.ts:29-66, 109-114`
|
||||
- `calculateDynamicMinHeight` has `chatHistory` in its dependency array, so it changes identity on every message update.
|
||||
- The `useEffect` at line 109 calls it on every `chatHistory` change.
|
||||
- It does `querySelector` + `getBoundingClientRect`, which forces browser layout recalculation (reflow).
|
||||
- Since `chatHistory` gets a new reference frequently (Finding #4), this triggers expensive layout recalculations very often.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
1. Debounce or throttle `calculateDynamicMinHeight` calls (e.g., only recalculate on user message additions, not during streaming).
|
||||
2. Decouple from `chatHistory` array identity — use `chatHistory.length` or a message count instead.
|
||||
3. Consider using `ResizeObserver` on the last message element rather than querying on every state change.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Every message update during streaming.
|
||||
- **Causes**: Layout thrashing (forced reflows) on the main thread.
|
||||
|
||||
---
|
||||
|
||||
## 6. [TODO] useAllNotes Sorts Entire File List on Every Vault Change (HIGH)
|
||||
|
||||
### Issue Description
|
||||
|
||||
The `useAllNotes` hook sorts all vault files by creation date inside `useMemo`, triggered on every debounced vault event.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/chat-components/hooks/useAllNotes.ts:36`
|
||||
- **Code**: `files.sort((a, b) => b.stat.ctime - a.stat.ctime)` runs inside `useMemo` with `[allNotes, isCopilotPlus]` deps.
|
||||
- `allNotes` atom gets a new array reference on every debounced vault event (`VaultDataManager.refreshNotes` at `vaultDataAtoms.ts:214` always sets a new array).
|
||||
- For vaults with 5000+ files, this is O(N log N) on every file create/delete/rename.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Move sorting into `VaultDataManager.refreshNotes()` so it happens once at the source, not in every consumer. Or pre-sort the atom value.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Users with large vaults (5000+ files).
|
||||
- **Triggers**: On every file create/delete/rename in the vault (debounced at 250ms).
|
||||
|
||||
---
|
||||
|
||||
## 7. [TODO] Loading Dots Animation Triggers Full ChatMessages Re-render Every 200ms (MEDIUM)
|
||||
|
||||
### Issue Description
|
||||
|
||||
The loading dots animation uses internal state (`setLoadingDots`) that triggers `ChatMessages` re-renders every 200ms, which cascades to all child message components.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/chat-components/ChatMessages.tsx:49-59`
|
||||
- A `setInterval` at 200ms calls `setLoadingDots()`, updating internal state of the `memo`-wrapped `ChatMessages`.
|
||||
- Internal state changes bypass `React.memo`, causing the entire message list to re-render.
|
||||
- Combined with Finding #1, all historical `ChatSingleMessage` children re-render too.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Extract the loading dots into a separate small component that manages its own state:
|
||||
|
||||
```tsx
|
||||
const LoadingDots: React.FC = () => {
|
||||
const [dots, setDots] = useState("");
|
||||
useEffect(() => {
|
||||
/* interval logic */
|
||||
}, []);
|
||||
return <span>{dots}</span>;
|
||||
};
|
||||
```
|
||||
|
||||
This isolates the 200ms re-renders to just the loading indicator, not the entire message list.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Every loading phase (waiting for AI response).
|
||||
- **Causes**: 5 unnecessary full-tree re-renders per second during loading.
|
||||
|
||||
---
|
||||
|
||||
## 8. [TODO] VaultDataManager Tag Refresh Scans All Markdown Files (MEDIUM)
|
||||
|
||||
### Issue Description
|
||||
|
||||
`refreshTagsFrontmatter()` and `refreshTagsAll()` each iterate over ALL markdown files, and both are triggered independently on every file modify/metadata change.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/state/vaultDataAtoms.ts:234-271`
|
||||
- Both methods call `app.vault.getMarkdownFiles()` and iterate with `getTagsFromNote()` on each file.
|
||||
- Both are triggered by `handleFileModify` and `handleMetadataChange` events (debounced at 250ms).
|
||||
- For a vault with 5000 markdown files, this means two full vault scans on every file save.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
1. Merge the two refresh methods into a single pass that computes both frontmatter and all tags simultaneously.
|
||||
2. Consider incremental tag updates — only recompute tags for the changed file, not the entire vault.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Users with large vaults.
|
||||
- **Triggers**: On every file save (after 250ms debounce).
|
||||
- **Usually mitigated by**: Debouncing. But for very large vaults, even one scan can take 50-100ms.
|
||||
|
||||
---
|
||||
|
||||
## 9. [TODO] Chat.tsx Callback Dependencies Include chatHistory Array (MEDIUM)
|
||||
|
||||
### Issue Description
|
||||
|
||||
`handleRegenerate`, `handleEdit`, and `handleDelete` in `Chat.tsx` all depend on `chatHistory` in their `useCallback` dependency arrays, causing them to be recreated on every render and defeating `ChatMessages`'s `React.memo`.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/Chat.tsx:421-472, 474-550, 664-683`
|
||||
- These callbacks access `chatHistory[messageIndex]` to get the message to operate on.
|
||||
- Since `chatHistory` gets a new array reference on every state update (Finding #4), these callbacks are recreated on every render.
|
||||
- They're passed as props to `ChatMessages` (which is memoized), but new callback references trigger re-renders regardless.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Use a ref to hold the latest `chatHistory` and access it inside the callbacks:
|
||||
|
||||
```typescript
|
||||
const chatHistoryRef = useRef(chatHistory);
|
||||
chatHistoryRef.current = chatHistory;
|
||||
|
||||
const handleRegenerate = useCallback(
|
||||
(messageIndex: number) => {
|
||||
const message = chatHistoryRef.current[messageIndex];
|
||||
// ... rest of logic
|
||||
},
|
||||
[
|
||||
/* stable deps only */
|
||||
]
|
||||
);
|
||||
```
|
||||
|
||||
This keeps the callback identity stable while always reading the latest data.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Effectively defeats the `React.memo` on `ChatMessages`, compounding with Finding #1.
|
||||
|
||||
---
|
||||
|
||||
## 10. [TODO] ChatSingleMessage DOM Manipulation in useEffect During Streaming (MEDIUM)
|
||||
|
||||
### Issue Description
|
||||
|
||||
The main rendering `useEffect` in `ChatSingleMessage` performs extensive synchronous DOM operations on every `message` prop change, which during streaming happens on every RAF tick.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/chat-components/ChatSingleMessage.tsx:533-723`
|
||||
- Operations include: `querySelectorAll`, `createElement`, `insertBefore`, `appendChild`, `remove`, `MarkdownRenderer.renderMarkdown()`.
|
||||
- During streaming, only the streaming `ChatSingleMessage` instance does this work (historical messages would too, per Finding #1, but they should not be receiving new props).
|
||||
- The `preprocess` callback (line 246) runs multiple regex replacements and string splitting on every update.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
1. Fix Finding #1 first — this eliminates DOM manipulation for historical messages during streaming.
|
||||
2. For the streaming message, consider differential updates (only re-render new content appended since last frame) rather than re-processing the entire message on every token.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Streaming message rendering.
|
||||
- **Mostly contained**: After fixing Finding #1, only one component instance does this work per frame.
|
||||
|
||||
---
|
||||
|
||||
## 11. [TODO] ChatSingleMessage preprocess: Repeated Regex Splitting (MEDIUM)
|
||||
|
||||
### Issue Description
|
||||
|
||||
The `replaceLinks` helper splits the message by code blocks using a regex, then runs further regex replacements on each part. This happens twice per render (once for images, once for links).
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/chat-components/ChatSingleMessage.tsx:378-395`
|
||||
- **Code**: `text.split(/(```[\s\S]*?```|`[^`]\*`)/g)` creates an array of code/non-code segments, then regex replacement runs on each non-code segment. Called twice in the preprocessing pipeline.
|
||||
- For long AI responses with many code blocks, this is O(parts x content_length) per call.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Split the content into code/non-code segments once, then apply all transformations to the non-code segments in a single pass.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Long AI responses during streaming.
|
||||
- **Severity scales with**: Message length and number of code blocks.
|
||||
|
||||
---
|
||||
|
||||
## 12. [TODO] Missing React.memo on Frequently-Rendered Leaf Components (LOW)
|
||||
|
||||
### Issue Description
|
||||
|
||||
Several leaf components that render frequently due to parent re-renders are not wrapped in `React.memo`.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **ChatButtons** (`src/components/chat-components/ChatButtons.tsx`): Renders for every message, receives callbacks that change on parent re-render.
|
||||
- **MessageContext** (`src/components/chat-components/ChatSingleMessage.tsx:85`): Renders context badges for each message.
|
||||
- **ChatHistoryItem** (`src/components/chat-components/ChatHistoryPopover.tsx:349`): Receives `confirmDeleteId` which changes for all items on any delete confirmation.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Wrap each in `React.memo`. For `ChatHistoryItem`, consider passing only a boolean `isConfirmingDelete` instead of the full `confirmDeleteId` to reduce unnecessary re-renders.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Individually negligible**, but compounds with other findings.
|
||||
|
||||
---
|
||||
|
||||
## 13. [TODO] useAtMentionSearch Eagerly Creates React Elements for All Vault Items (LOW)
|
||||
|
||||
### Issue Description
|
||||
|
||||
The `noteItems`, `folderItems`, and `webTabItems` memos create `React.createElement` for icon components on every item, even when the typeahead menu is not open.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/chat-components/hooks/useAtMentionSearch.ts:45-113`
|
||||
- For vaults with 5000+ notes, this creates 5000+ React elements on mount.
|
||||
- The `useMemo` deps include `allNotes` which changes on every vault event.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Defer icon element creation to render time (pass icon component type instead of element instance), or only compute items when the typeahead is open.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Affects**: Initial mount time and memory in large vaults.
|
||||
- **Mitigated by**: `useMemo` — only recomputes when deps change.
|
||||
|
||||
---
|
||||
|
||||
## 14. [TODO] ChatHistoryPopover ChatHistoryItem Not Memoized (LOW)
|
||||
|
||||
### Issue Description
|
||||
|
||||
`ChatHistoryItem` receives several props that change across all items when any single item is being edited or deleted.
|
||||
|
||||
### Technical Details
|
||||
|
||||
- **File**: `src/components/chat-components/ChatHistoryPopover.tsx:349-486`
|
||||
- `confirmDeleteId` changes for all items when any delete is confirmed.
|
||||
- `editingTitle` changes on every keystroke during editing.
|
||||
- All items re-render when either of these change.
|
||||
|
||||
### Recommended Solution
|
||||
|
||||
Wrap `ChatHistoryItem` in `React.memo`. Pass derived booleans (`isConfirmingDelete`, `isEditing`) instead of global IDs.
|
||||
|
||||
### Impact
|
||||
|
||||
- **Mitigated by**: Pagination (max 50 items rendered at a time).
|
||||
- **Nearly negligible** with current architecture.
|
||||
|
||||
---
|
||||
|
||||
## Good Patterns Already in Place
|
||||
|
||||
These patterns were identified as already well-implemented:
|
||||
|
||||
- **RAF-throttled streaming**: `useRafThrottledCallback` properly throttles streaming text updates to animation frames
|
||||
- **ChatMessages memo**: The top-level `ChatMessages` is wrapped in `React.memo` (though currently defeated by unstable props — see Findings #1, #4, #9)
|
||||
- **Pagination in ChatHistoryPopover**: IntersectionObserver-based infinite scroll prevents rendering all history at once
|
||||
- **VaultDataManager debouncing**: 250ms debounce on vault events prevents rapid-fire re-scans
|
||||
- **useLayoutEffect for pagination reset**: Prevents one-frame render spike when popover opens
|
||||
- **RelevantNotes memo**: Properly memoized with `React.memo`
|
||||
- **Streaming message isolation**: The streaming message uses a separate `ChatSingleMessage` instance with a stable key, preventing full list re-keying
|
||||
|
||||
---
|
||||
|
||||
## Recommended Fix Priority Order
|
||||
|
||||
| Priority | Finding | Effort | Expected Impact |
|
||||
| -------- | -------------------------------------------------- | ---------- | ------------------------------------------ |
|
||||
| P0 | #1 Memoize ChatSingleMessage + stabilize callbacks | Medium | Eliminates streaming jank |
|
||||
| P0 | #2 Hoist filter out of .map() | Trivial | O(N^2) -> O(N) per render |
|
||||
| P1 | #9 Stabilize Chat.tsx callback dependencies | Small | Unbreaks ChatMessages memo |
|
||||
| P1 | #4 Stabilize chatHistory array reference | Small | Prevents cascading re-renders |
|
||||
| P1 | #7 Extract loading dots component | Trivial | Eliminates 5 re-renders/sec during loading |
|
||||
| P2 | #3 Cache ChatPersistenceManager | Trivial | Reduces GC pressure |
|
||||
| P2 | #5 Decouple scroll calculation from chatHistory | Small | Eliminates layout thrashing |
|
||||
| P2 | #6 Pre-sort notes in VaultDataManager | Trivial | Reduces sort cost in large vaults |
|
||||
| P3 | #8 Merge tag refresh into single pass | Small | Halves vault scan frequency |
|
||||
| P3 | #10-14 Remaining medium/low items | Small each | Incremental improvements |
|
||||
|
||||
---
|
||||
|
||||
_Last updated: 2026-03-03_
|
||||
174
docs/agent-mode-and-tools.md
Normal file
|
|
@ -0,0 +1,174 @@
|
|||
# Agent Mode and Tools
|
||||
|
||||
Copilot Plus includes an **autonomous agent** that can reason step-by-step and decide which tools to use to answer your question. Instead of you specifying every step, the agent figures out what to do on its own.
|
||||
|
||||
This feature requires a [Copilot Plus](copilot-plus-and-self-host.md) license.
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
When the autonomous agent is enabled, Copilot can:
|
||||
|
||||
1. Break down your request into sub-tasks
|
||||
2. Use tools to gather information (search your vault, search the web, read a note)
|
||||
3. Create or edit notes
|
||||
4. Combine results and give you a comprehensive answer
|
||||
|
||||
**Example**: Ask "What did I work on last week?" and the agent will automatically search your vault for dated notes from the past 7 days, read the relevant ones, and summarize your week.
|
||||
|
||||
---
|
||||
|
||||
## Enabling Agent Mode
|
||||
|
||||
1. Go to **Settings → Copilot → Plus**
|
||||
2. Turn on **Enable Autonomous Agent**
|
||||
|
||||
The agent activates automatically when you're in **Copilot Plus** mode. You don't need to do anything special — just ask your question.
|
||||
|
||||
### Max Iterations
|
||||
|
||||
The agent works in iteration cycles (think → use a tool → think → use a tool → answer). You can control the maximum number of iterations before the agent stops:
|
||||
|
||||
- **Default**: 4 iterations
|
||||
- **Maximum**: 16 iterations
|
||||
- **Setting**: **Settings → Copilot → Plus → Autonomous Agent Max Iterations**
|
||||
|
||||
The agent also has a maximum runtime of 5 minutes per response, regardless of iteration count.
|
||||
|
||||
---
|
||||
|
||||
## Available Tools
|
||||
|
||||
Copilot Plus has 13 built-in tools. Some are always active; others can be enabled or disabled.
|
||||
|
||||
### Always-Enabled Tools
|
||||
|
||||
These tools are always available and cannot be disabled:
|
||||
|
||||
#### Get Current Time
|
||||
Gets the current time in any timezone. Useful for time-aware queries like "what should I do today?"
|
||||
|
||||
#### Get Time Range
|
||||
Converts natural time expressions (like "last week" or "yesterday") into exact date ranges. Usually called automatically before a time-based vault search.
|
||||
|
||||
#### Get Time Info
|
||||
Converts an epoch timestamp to a human-readable date and time.
|
||||
|
||||
#### Convert Timezones
|
||||
Converts a time from one timezone to another. Ask: "What time is 3pm EST in Tokyo?"
|
||||
|
||||
#### Read Note
|
||||
Reads the content of a specific note. The agent uses this to inspect a note it found via search, or that you mentioned explicitly. Works on large notes by reading them in chunks.
|
||||
|
||||
#### File Tree
|
||||
Browses the file structure of your vault. The agent uses this to find folder paths before creating new notes or to count files in a folder.
|
||||
|
||||
#### Tag List
|
||||
Lists all tags in your vault with usage statistics. Useful for tag reorganization or finding notes by tag patterns.
|
||||
|
||||
#### Update Memory
|
||||
Saves information to your memory when you explicitly ask the AI to remember something. See [Copilot Plus and Self-Host](copilot-plus-and-self-host.md#memory-system) for details.
|
||||
|
||||
> **Requires**: **Settings → Copilot → Plus → Reference Saved Memories** must be enabled. If this setting is off, the tool is not registered and memory commands will not work.
|
||||
|
||||
### Configurable Tools
|
||||
|
||||
These tools can be individually enabled or disabled in **Settings → Copilot → Plus → Tool Settings**:
|
||||
|
||||
#### Vault Search
|
||||
Searches your vault notes by content. The agent uses this to find notes relevant to your question.
|
||||
|
||||
- **Trigger**: Automatically for vault-related questions, or explicitly with `@vault`
|
||||
- **Uses**: Both semantic search (if enabled) and lexical search
|
||||
|
||||
#### Web Search
|
||||
Searches the internet for current information.
|
||||
|
||||
- **Trigger**: Automatically when your question implies web/online content, or explicitly with `@websearch` or `@web`
|
||||
- **Requires**: A web search service configured (Firecrawl or Perplexity in self-host mode, or handled by Plus)
|
||||
|
||||
#### Write to File
|
||||
Creates a new note or overwrites an existing one entirely.
|
||||
|
||||
- **Trigger**: Automatically for "create a note" requests, or explicitly with `@composer` (available in both Copilot Plus and Projects mode)
|
||||
- **Behavior**: Shows a preview of the content before writing. You can review and accept or reject the change.
|
||||
- **Auto-accept**: Enable **Settings → Copilot → Plus → Auto-accept edits** to skip the preview
|
||||
|
||||
#### Replace in File
|
||||
Makes targeted changes to an existing note using search-and-replace blocks.
|
||||
|
||||
- **Use case**: Small edits (adding a bullet, updating a section) — more precise than rewriting the whole note
|
||||
- **Behavior**: Shows a diff preview before applying the change
|
||||
- **Auto-accept**: Same setting as Write to File
|
||||
|
||||
#### YouTube Transcription
|
||||
Fetches the transcript of a YouTube video.
|
||||
|
||||
- **Trigger**: Automatically when you paste a YouTube URL in your message
|
||||
- **No extra setup needed**: Just include the URL in your message
|
||||
- **Self-host option**: Use your own Supadata API key for transcription in self-host mode
|
||||
|
||||
---
|
||||
|
||||
## Tool Settings
|
||||
|
||||
Go to **Settings → Copilot → Plus → Tool Settings** to:
|
||||
- See all available tools
|
||||
- Enable or disable individual configurable tools
|
||||
- View what each tool does
|
||||
|
||||
---
|
||||
|
||||
## Using Tools Explicitly
|
||||
|
||||
While the agent automatically decides when to use tools, you can also trigger them explicitly with @-mentions:
|
||||
|
||||
```
|
||||
@vault find all notes about my reading list
|
||||
@websearch what is the latest version of Python?
|
||||
@composer create a new meeting notes template
|
||||
@memory remember that I prefer bullet points for lists
|
||||
```
|
||||
|
||||
See [Context and Mentions](context-and-mentions.md) for the full @-mention reference.
|
||||
|
||||
---
|
||||
|
||||
## Tool Call Indicators
|
||||
|
||||
While the agent is working, the chat shows status indicators for each tool call:
|
||||
- "Reading files"
|
||||
- "Searching the web"
|
||||
- "Reading file tree"
|
||||
- "Compacting"
|
||||
|
||||
This lets you see what the agent is doing as it works.
|
||||
|
||||
---
|
||||
|
||||
## File Editing: Preview and Diff
|
||||
|
||||
When the agent uses **Write to File** or **Replace in File**, it shows a preview before making changes:
|
||||
|
||||
- **Split view**: Before/after shown side by side
|
||||
- **Side-by-side view**: Changes highlighted inline
|
||||
|
||||
You can choose your preferred diff view in **Settings → Copilot → Plus → Diff View Mode**.
|
||||
|
||||
Review the proposed change and click:
|
||||
- **Accept** — Apply the change to your note
|
||||
- **Reject** — Discard without making any changes
|
||||
- **Revert** — Undo a change that was already accepted
|
||||
|
||||
### Auto-Accept Edits
|
||||
|
||||
If you trust the agent and don't want to review every file change, enable **Auto-accept edits** in **Settings → Copilot → Plus**. File changes will be applied immediately without a confirmation step.
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Copilot Plus and Self-Host](copilot-plus-and-self-host.md) — Licensing and memory
|
||||
- [Vault Search and Indexing](vault-search-and-indexing.md) — How vault search works
|
||||
- [Context and Mentions](context-and-mentions.md) — @-mention triggers for tools
|
||||
159
docs/chat-interface.md
Normal file
|
|
@ -0,0 +1,159 @@
|
|||
# Chat Interface
|
||||
|
||||
The Copilot chat panel is the main way you interact with AI in Obsidian. This guide covers everything about the chat UI: modes, message controls, history, settings, and advanced features like auto-compact.
|
||||
|
||||
---
|
||||
|
||||
## Chat Modes
|
||||
|
||||
Copilot offers four modes. You can switch between them using the mode selector at the top of the chat panel.
|
||||
|
||||
### Chat
|
||||
General-purpose conversation. Good for writing, brainstorming, summarizing, or any task where you want to talk to an AI. Your currently open note and selected text are automatically included as context.
|
||||
|
||||
### Vault QA (Basic)
|
||||
Ask questions about your vault content. Copilot uses lexical search (keyword matching) to find relevant notes and passes them as context to the AI. No indexing required. Good for quick questions about your notes.
|
||||
|
||||
### Copilot Plus
|
||||
The most powerful mode. Requires a [Copilot Plus](copilot-plus-and-self-host.md) license. Combines Chat and Vault QA with an autonomous agent that can:
|
||||
- Search your vault and the web
|
||||
- Read and edit notes
|
||||
- Remember things across conversations
|
||||
- Use a growing set of tools automatically
|
||||
|
||||
### Projects (alpha)
|
||||
Focused workspaces with their own context, model, system prompt, and isolated chat history. Useful for keeping separate AI conversations per project. See [Projects](projects.md) for details.
|
||||
|
||||
---
|
||||
|
||||
## Sending Messages
|
||||
|
||||
Type your message in the input box at the bottom of the chat panel and press **Enter** to send (or **Shift+Enter** to add a new line). You can change the send key in Settings → Basic → **Default Send Shortcut**.
|
||||
|
||||
While the AI is generating a response, a **Stop** button appears. Click it to interrupt the stream at any time.
|
||||
|
||||
### Referencing Notes Inline
|
||||
|
||||
You can mention specific notes directly in your message using double-bracket syntax:
|
||||
|
||||
```
|
||||
[[Note Title]]
|
||||
```
|
||||
|
||||
Copilot adds the note's content to your message as context in the background. This is different from @-mentions — it's typed directly in your message text.
|
||||
|
||||
### User Message Buttons
|
||||
|
||||
Each message you send has action buttons that appear on hover:
|
||||
- **Edit** — Modify your prompt. Press Enter to re-send the edited message to the AI.
|
||||
- **Copy** — Copy the message text to clipboard
|
||||
- **Delete** — Remove this message from the conversation
|
||||
|
||||
### AI Message Buttons
|
||||
|
||||
Each AI response has action buttons:
|
||||
- **Insert at cursor** — Insert the AI's response at your cursor position in the active note
|
||||
- **Replace at cursor** — Replace the selected text in your note with the AI's response
|
||||
- **Copy** — Copy the response to clipboard
|
||||
- **Regenerate** — Ask the AI to generate a new response to the same message
|
||||
- **Delete** — Remove this response from the conversation
|
||||
|
||||
---
|
||||
|
||||
## Chat History
|
||||
|
||||
### Autosave
|
||||
|
||||
By default, Copilot automatically saves your conversations as markdown files in your vault. Each saved chat appears in the `copilot/copilot-conversations/` folder.
|
||||
|
||||
You can turn off autosave in Settings → Basic. When you start a new chat, any unsaved conversation is saved automatically.
|
||||
|
||||
### Chat File Name Format
|
||||
|
||||
The filename template controls how saved chats are named. The default is:
|
||||
|
||||
```
|
||||
{$topic}@{$date}_{$time}
|
||||
```
|
||||
|
||||
Where:
|
||||
- `{$topic}` — An AI-generated title (or the first few words of your first message if AI titles are off)
|
||||
- `{$date}` — Date in YYYY-MM-DD format
|
||||
- `{$time}` — Time in HH-MM-SS format
|
||||
|
||||
All three variables are required. You can customize the format in Settings → Basic → **Conversation note name**.
|
||||
|
||||
### AI-Generated Titles
|
||||
|
||||
When **Generate AI chat title on save** is enabled (default), Copilot asks the AI to generate a short, descriptive title for the conversation when saving. When disabled, the first 10 words of your first message are used instead.
|
||||
|
||||
### Loading Previous Chats
|
||||
|
||||
Click the **clock/history icon** in the chat panel toolbar to open the Chat History list. You can:
|
||||
- Browse previous conversations
|
||||
- Click a conversation to load it and continue from where you left off
|
||||
- Delete conversations you no longer need
|
||||
|
||||
The history list can be sorted by most recent or alphabetically.
|
||||
|
||||
---
|
||||
|
||||
## Per-Session Settings (Gear Icon)
|
||||
|
||||
Click the **gear icon** inside the chat panel to open per-session settings. These apply only to the current conversation and reset when you start a new chat:
|
||||
|
||||
- **System prompt** — Override the default system prompt for this session
|
||||
- **Temperature** — Controls randomness (0 = deterministic, 1 = creative)
|
||||
- **Max tokens** — Maximum length of the AI's response
|
||||
|
||||
---
|
||||
|
||||
## Token Counter
|
||||
|
||||
Copilot shows a token count indicator at the bottom of the chat. This estimates how many tokens are being used by your current context. Useful for knowing when you're approaching context limits.
|
||||
|
||||
---
|
||||
|
||||
## Auto-Compact
|
||||
|
||||
When a conversation grows very long, it can exceed the model's context window. Auto-compact automatically summarizes the older portion of the conversation and replaces it with a compressed summary, letting you continue chatting without losing track of what was discussed.
|
||||
|
||||
The threshold is configured in Settings → Basic → **Auto-compact threshold**, which defaults to 128,000 tokens. Valid range: 64,000–1,000,000 tokens.
|
||||
|
||||
When auto-compact triggers, you'll see a "Compacting" indicator in the chat. The conversation continues normally — older messages are replaced by a summary, so the AI still understands the history even though you can no longer scroll back to see the original messages.
|
||||
|
||||
---
|
||||
|
||||
## Suggested Prompts
|
||||
|
||||
When starting a new chat, Copilot may show suggested prompts based on your active note or previous conversations. You can enable or disable this in Settings → Basic → **Show suggested prompts**.
|
||||
|
||||
## Relevant Notes
|
||||
|
||||
Copilot can display a list of notes related to your currently active note in the chat panel. This helps surface notes you might want to reference without manually searching.
|
||||
|
||||
Enable in **Settings → Copilot → Basic → Relevant Notes** (on by default).
|
||||
|
||||
## Saving a Chat Manually
|
||||
|
||||
If autosave is off, or you want to save mid-conversation, click the **Save Chat as Note** button above the chat input box. This saves the current conversation to your configured save folder.
|
||||
|
||||
---
|
||||
|
||||
## New Chat Behavior
|
||||
|
||||
Click the **pencil/new chat icon** to start a fresh conversation. This:
|
||||
1. Saves the current conversation (if autosave is enabled)
|
||||
2. Clears the chat window
|
||||
3. Resets the context to your currently active note
|
||||
|
||||
You can also use the command palette: **New Copilot Chat**.
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Context and Mentions](context-and-mentions.md) — Control what context the AI sees
|
||||
- [System Prompts](system-prompts.md) — Customize AI behavior with system prompts
|
||||
- [Agent Mode and Tools](agent-mode-and-tools.md) — What Plus mode can do
|
||||
- [Projects](projects.md) — Isolated workspaces with separate histories
|
||||
145
docs/context-and-mentions.md
Normal file
|
|
@ -0,0 +1,145 @@
|
|||
# Context and Mentions
|
||||
|
||||
Copilot uses **context** to give the AI information about your notes, selected text, web content, and more. You can control exactly what context the AI sees using automatic context, @-mentions, and manual commands.
|
||||
|
||||
---
|
||||
|
||||
## Automatic Context
|
||||
|
||||
### Active Note
|
||||
|
||||
By default, the content of your currently open note is automatically included in every message you send. This means you can ask things like:
|
||||
|
||||
- "Summarize this note"
|
||||
- "What are the action items here?"
|
||||
- "Add a conclusion section"
|
||||
|
||||
To disable automatic note context: **Settings → Copilot → Basic → Auto-add active note to context** (toggle off).
|
||||
|
||||
### Active Web Tab (Desktop Only)
|
||||
|
||||
If you have the Copilot Web Viewer open alongside your notes, the content of the currently active web tab is automatically included as context (labeled `{activeWebTab}`). This lets you ask the AI to help you work with web content.
|
||||
|
||||
### Selected Text
|
||||
|
||||
If you highlight text in a note and then type in the chat, the selected text is automatically included as context. This is useful for asking about or transforming a specific part of a note.
|
||||
|
||||
You can enable/disable automatic selection adding in **Settings → Copilot → Basic → Auto-add selection to context**.
|
||||
|
||||
### Images in Markdown
|
||||
|
||||
If your note contains images (e.g., `![[screenshot.png]]`), and you're using a model with **Vision** capability, those images are automatically included in the context. Copilot will pass the image data to the AI so it can see and describe the image.
|
||||
|
||||
To control this behavior: **Settings → Copilot → Basic → Pass markdown images to AI**.
|
||||
|
||||
---
|
||||
|
||||
## @-Mentions
|
||||
|
||||
Type `@` in the chat input to mention and include specific items as context.
|
||||
|
||||
### @note — Include a Specific Note
|
||||
|
||||
Type `@` followed by the note title to add a note to context:
|
||||
|
||||
```
|
||||
@My Meeting Notes tell me what was decided in this meeting
|
||||
```
|
||||
|
||||
The note's full content is included in the request.
|
||||
|
||||
### @folder — Include a Folder of Notes
|
||||
|
||||
Type `@` followed by a folder name to include all notes in that folder:
|
||||
|
||||
```
|
||||
@Projects/ what tasks are still open?
|
||||
```
|
||||
|
||||
### @tags — Include Notes by Tag
|
||||
|
||||
Use `#` after `@` to include all notes with a specific tag:
|
||||
|
||||
```
|
||||
@#work/project summarize the status of the work project
|
||||
```
|
||||
|
||||
### @URL — Include a Web Page
|
||||
|
||||
Paste a URL or type `@https://...` to fetch and include a web page's content:
|
||||
|
||||
```
|
||||
@https://example.com/article summarize this article
|
||||
```
|
||||
|
||||
URL processing requires Copilot Plus. YouTube URLs are handled specially — Copilot will fetch the video transcript automatically.
|
||||
|
||||
### Tool Mentions
|
||||
|
||||
These special @-mentions explicitly trigger tools in Copilot Plus mode:
|
||||
|
||||
| Mention | What it does |
|
||||
|---|---|
|
||||
| `@vault` | Search your vault notes for relevant information |
|
||||
| `@websearch` or `@web` | Search the internet |
|
||||
| `@composer` | Create or edit a note |
|
||||
| `@memory` | Access or update your memory |
|
||||
|
||||
Example:
|
||||
```
|
||||
@vault what did I write about machine learning last month?
|
||||
@websearch what are the latest changes to the Python packaging ecosystem?
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Adding Context Manually
|
||||
|
||||
### Add Selection to Chat Context
|
||||
|
||||
Use the command palette: **Add selection to chat context**
|
||||
|
||||
Highlights the selected text and adds it to the chat as context without sending a message. Useful when you want to build up context before sending.
|
||||
|
||||
### Add Web Selection to Chat Context
|
||||
|
||||
Use the command palette: **Add web selection to chat context**
|
||||
|
||||
Works similarly but captures selected text from the Web Viewer. Available on desktop only.
|
||||
|
||||
### Adding a PDF as Context (Copilot Plus)
|
||||
|
||||
Click the **+ Add context** button above the chat input to attach a PDF file. The PDF is converted to text and included as context for your message.
|
||||
|
||||
### Adding an Image as Context
|
||||
|
||||
Drag an image directly into the chat input box, or click the **image button** in the bottom-right corner of the chat input. The image is sent to the AI if your selected model supports **Vision** capability.
|
||||
|
||||
---
|
||||
|
||||
## Context Indicators
|
||||
|
||||
When context items are added to your message, Copilot shows small pills or badges in the chat input area showing what's included (e.g., the note name, a URL, a tag). This helps you confirm exactly what the AI will see.
|
||||
|
||||
---
|
||||
|
||||
## Context Behavior by Mode
|
||||
|
||||
| Context Type | Chat | Vault QA | Copilot Plus |
|
||||
|---|---|---|---|
|
||||
| Active note | Yes (auto) | Yes (auto) | Yes (auto) |
|
||||
| Selected text | Yes (auto) | Yes (auto) | Yes (auto) |
|
||||
| @note / @folder | Yes | Yes | Yes |
|
||||
| @URL processing | Copilot Plus only | Copilot Plus only | Yes |
|
||||
| @vault search | Yes (explicit) | Auto | Auto |
|
||||
| @websearch | No | No | Yes |
|
||||
| Images (vision) | Yes | Yes | Yes |
|
||||
| Active web tab | Desktop only | Desktop only | Desktop only |
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Chat Interface](chat-interface.md) — How the chat panel works
|
||||
- [Agent Mode and Tools](agent-mode-and-tools.md) — More on @vault and @websearch
|
||||
- [Vault Search and Indexing](vault-search-and-indexing.md) — How vault search works
|
||||
157
docs/copilot-plus-and-self-host.md
Normal file
|
|
@ -0,0 +1,157 @@
|
|||
# Copilot Plus and Self-Host
|
||||
|
||||
**Copilot Plus** is a premium tier that unlocks advanced features beyond the free, API-key-based experience. **Self-Host Mode** is an additional option for Copilot Plus Lifetime/Believer subscribers who want to run their own infrastructure.
|
||||
|
||||
---
|
||||
|
||||
## Copilot Plus
|
||||
|
||||
### What Is Copilot Plus?
|
||||
|
||||
Copilot Plus is a subscription that enables:
|
||||
|
||||
- **Autonomous agent mode** — AI that reasons step-by-step and uses tools automatically
|
||||
- **File editing tools** — Write to File and Replace in File for AI-driven note editing
|
||||
- **Web search** — Search the internet from chat
|
||||
- **YouTube transcription** — Fetch video transcripts and use them as context
|
||||
- **Memory system** — Persistent memory across conversations
|
||||
- **Copilot Plus Flash model** — A built-in model that requires no separate API key
|
||||
- **URL processing** — Fetch and summarize web pages as context
|
||||
- **Copilot Plus embedding models** — High-quality embeddings for semantic search
|
||||
|
||||
### Setting Up Copilot Plus
|
||||
|
||||
1. Get a license key from your dashboard at **https://www.obsidiancopilot.com/en/dashboard**
|
||||
2. Go to **Settings → Copilot → Basic** (or the Plus banner in the settings)
|
||||
3. Enter your license key in the **Copilot Plus License Key** field
|
||||
4. Features unlock automatically
|
||||
|
||||
---
|
||||
|
||||
## Copilot Plus Flash Model
|
||||
|
||||
**Copilot Plus Flash** is a built-in AI model included with your Copilot Plus subscription:
|
||||
|
||||
- No separate API key needed
|
||||
- Works out of the box once your license key is active
|
||||
- Supports vision (image inputs)
|
||||
- Good for general-purpose tasks
|
||||
|
||||
It appears as `copilot-plus-flash` in the model selector.
|
||||
|
||||
---
|
||||
|
||||
## Memory System
|
||||
|
||||
The memory system lets Copilot remember things across conversations, so you don't have to repeat yourself.
|
||||
|
||||
### Recent Conversations
|
||||
|
||||
Copilot can reference your recent conversation history to provide more contextually relevant responses. This is separate from the current chat window — it's a summary of what you've been working on.
|
||||
|
||||
- **Enable**: **Settings → Copilot → Plus → Reference Recent Conversation** (on by default)
|
||||
- **How many**: **Settings → Copilot → Plus → Max Recent Conversations** — default 30, range 10–50
|
||||
- All history is stored locally in your vault (no data leaves your machine for this feature)
|
||||
|
||||
### Saved Memories
|
||||
|
||||
You can ask Copilot to explicitly remember specific facts about you:
|
||||
|
||||
```
|
||||
@memory remember that I'm preparing for JLPT N3 and prefer bullet-point summaries
|
||||
```
|
||||
|
||||
Copilot saves this to a memory file in your vault and references it in future conversations.
|
||||
|
||||
- **Enable**: **Settings → Copilot → Plus → Reference Saved Memories** (on by default)
|
||||
- **Memory folder**: **Settings → Copilot → Plus → Memory Folder Name** — default: `copilot/memory`
|
||||
- **Update memory tool**: The AI can add, update, or remove memories when you ask
|
||||
|
||||
---
|
||||
|
||||
## Document Processor
|
||||
|
||||
When Copilot processes PDFs and other non-markdown files (in Plus mode), it converts them to markdown for the AI to read.
|
||||
|
||||
You can optionally save the converted markdown to a folder in your vault:
|
||||
|
||||
- **Setting**: **Settings → Copilot → Plus → Store converted markdown at**
|
||||
- Leave empty to skip saving (conversion still happens, it just isn't persisted)
|
||||
|
||||
---
|
||||
|
||||
## Self-Host Mode
|
||||
|
||||
### What Is Self-Host Mode?
|
||||
|
||||
Self-Host Mode lets you replace Copilot's cloud services with your own infrastructure. Instead of relying on Copilot's Plus backend, you run everything locally or on your own server.
|
||||
|
||||
**Requires**: A Copilot Plus Lifetime or Believer license (not available on monthly subscriptions).
|
||||
|
||||
### What Self-Host Mode Enables
|
||||
|
||||
- Use local or custom LLM servers
|
||||
- Custom web search via Firecrawl or Perplexity Sonar
|
||||
- Local YouTube transcript extraction via Supadata
|
||||
- Miyo desktop app for local PDF parsing, semantic search, and more
|
||||
|
||||
### Enabling Self-Host Mode
|
||||
|
||||
1. Go to **Settings → Copilot → Plus**
|
||||
2. Under **Self-Host Mode**, toggle **Enable Self-Host Mode**
|
||||
3. Copilot validates your license. If valid, the toggle activates.
|
||||
4. Toggle **Enable Miyo** to use the Miyo desktop app for local search, PDF parsing, and context.
|
||||
5. *(Optional)* Set **Custom Miyo Server URL** only if Miyo is running on a remote machine. Leave blank to use automatic local service discovery.
|
||||
|
||||
### Web Search in Self-Host Mode
|
||||
|
||||
Choose your web search provider:
|
||||
|
||||
- **Firecrawl** — A web crawling and scraping API. Get a key at firecrawl.dev. Enter it in **Settings → Copilot → Plus → Firecrawl API Key**.
|
||||
- **Perplexity Sonar** — An AI-powered search API. Get a key at perplexity.ai. Enter it in **Settings → Copilot → Plus → Perplexity API Key**.
|
||||
|
||||
### YouTube Transcription in Self-Host Mode
|
||||
|
||||
Use your own Supadata API key for YouTube transcript extraction:
|
||||
|
||||
- Get a key at supadata.ai
|
||||
- Enter it in **Settings → Copilot → Plus → Supadata API Key**
|
||||
|
||||
---
|
||||
|
||||
## Miyo Desktop App
|
||||
|
||||
Miyo is a companion desktop app from the same developer that enhances Copilot with local, offline capabilities:
|
||||
|
||||
### What Miyo Provides
|
||||
|
||||
- **Local semantic search** — Fast vector search without embedding API calls
|
||||
- **PDF parsing** — Converts PDFs to markdown locally (no cloud OCR)
|
||||
- **Context hub** — Manages your indexed documents locally
|
||||
- **Custom server URL** — Run Miyo on any machine (local or server)
|
||||
|
||||
### Setting Up Miyo
|
||||
|
||||
1. Download and install the Miyo desktop app
|
||||
2. Start the Miyo server
|
||||
3. In Copilot, go to **Settings → Copilot → Plus → Enable Miyo Search**
|
||||
4. Miyo automatically connects to the local server (or use a custom URL in **Miyo Server URL**)
|
||||
5. Index your vault — Copilot will use Miyo to generate and store embeddings locally
|
||||
|
||||
### Custom Miyo Server URL
|
||||
|
||||
If Miyo is running on a different machine (e.g., a home server), enter its address:
|
||||
|
||||
```
|
||||
http://192.168.1.10:8742
|
||||
```
|
||||
|
||||
Leave empty to use automatic local discovery.
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Agent Mode and Tools](agent-mode-and-tools.md) — Using the autonomous agent
|
||||
- [Vault Search and Indexing](vault-search-and-indexing.md) — How Miyo enhances semantic search
|
||||
- [Getting Started](getting-started.md) — First-time setup
|
||||
154
docs/custom-commands.md
Normal file
|
|
@ -0,0 +1,154 @@
|
|||
# Custom Commands
|
||||
|
||||
Custom commands are preset AI prompts you define once and reuse on any note or selected text. They're stored as markdown files in your vault and can be triggered from the right-click context menu, the command palette, or as slash commands in chat.
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
A custom command is like a template prompt. You write an instruction (with optional variables) and save it. From then on, you can apply it to any note or selected text with a single click.
|
||||
|
||||
**Examples of what you might create:**
|
||||
- "Summarize this note in bullet points"
|
||||
- "Extract all action items as a task list"
|
||||
- "Rewrite this in a more formal tone"
|
||||
- "Translate to Spanish"
|
||||
- "Create a Fleeting Note from this"
|
||||
|
||||
---
|
||||
|
||||
## Creating a Custom Command
|
||||
|
||||
### From Settings
|
||||
|
||||
1. Go to **Settings → Copilot → Command**
|
||||
2. Click **Add new command**
|
||||
3. Fill in the fields:
|
||||
- **Name** — What the command is called (also becomes its ID)
|
||||
- **Prompt** — The instruction to send to the AI
|
||||
- **Show in context menu** — Whether it appears when right-clicking text in a note
|
||||
- **Model** — Optional: use a specific model for this command (defaults to the current chat model)
|
||||
4. Save
|
||||
|
||||
### From the Command Palette
|
||||
|
||||
You can also create a command on the fly:
|
||||
|
||||
1. Open the command palette (`Ctrl/Cmd+P`)
|
||||
2. Run **Add new custom command**
|
||||
3. A form will open to fill in the command details
|
||||
|
||||
---
|
||||
|
||||
## Prompt Template Variables
|
||||
|
||||
Inside your prompt, you can use variables that get replaced with real content when the command runs:
|
||||
|
||||
| Variable | What it inserts |
|
||||
|---|---|
|
||||
| `{}` or `{selected_text}` | The text currently selected in the editor |
|
||||
| `{activeNote}` | The full content of the currently active note |
|
||||
| `{[[Note Title]]}` | The content of a specific note by title |
|
||||
| `{FolderPath}` | All notes within a specific folder |
|
||||
| `{#tag1, #tag2}` | All notes with any of the specified tags |
|
||||
|
||||
> **Important**: Tags in `{#tag1, #tag2}` must be in the note's **properties (frontmatter)**, not inline tags within the note body.
|
||||
|
||||
**Example — quiz generator using two variables:**
|
||||
```
|
||||
Come up with multiple choice questions using {activeNote}, and follow
|
||||
the format of {[[Quiz Template]]} to start a quiz session.
|
||||
|
||||
Ask one question at a time, stop and wait for the user.
|
||||
After the user answers, provide the correct answer and explanation.
|
||||
Repeat until the user says STOP.
|
||||
```
|
||||
|
||||
**Example — comparison using specific notes:**
|
||||
```
|
||||
Compare my notes on {[[Product Roadmap]]} and {[[Competitor Analysis]]} and identify gaps.
|
||||
```
|
||||
|
||||
**Example — acting on selected text:**
|
||||
```
|
||||
Rewrite this in a more formal tone: {selected_text}
|
||||
```
|
||||
|
||||
Variable substitution must be enabled in **Settings → Copilot → Command → Enable custom prompt templating** (on by default).
|
||||
|
||||
---
|
||||
|
||||
## Using Commands
|
||||
|
||||
### From the Right-Click Context Menu
|
||||
|
||||
If a command has **Show in context menu** enabled:
|
||||
1. Select some text in a note (optional)
|
||||
2. Right-click to open the context menu
|
||||
3. Hover over **Copilot** → select your command
|
||||
4. The AI processes your selection or note and shows the result
|
||||
|
||||
### From the Command Palette
|
||||
|
||||
1. Select text or open the note you want to work with
|
||||
2. Open the command palette (`Ctrl/Cmd+P`)
|
||||
3. Run **Apply custom command**
|
||||
4. Pick your command from the list
|
||||
|
||||
### As a Slash Command in Chat
|
||||
|
||||
Inside the chat input, type `/` followed by the command name to run it:
|
||||
|
||||
```
|
||||
/summarize
|
||||
```
|
||||
|
||||
The command runs in the context of your current chat session and active note.
|
||||
|
||||
> **Note**: The `@composer` mention (for AI note editing) requires Copilot Plus. In free modes, `@composer` will not be available.
|
||||
|
||||
---
|
||||
|
||||
## Managing Commands
|
||||
|
||||
Go to **Settings → Copilot → Command** to manage all your custom commands:
|
||||
|
||||
- **Edit** — Click the edit icon next to any command
|
||||
- **Reorder** — Drag commands to change their order (affects the context menu and command list)
|
||||
- **Duplicate** — Copy an existing command as a starting point
|
||||
- **Delete** — Remove a command permanently
|
||||
- **Sort strategy** — Choose how commands are sorted: manually, by recent use, or alphabetically
|
||||
|
||||
### Custom Prompts Folder
|
||||
|
||||
Commands are stored as markdown files in your vault. The default folder is `copilot/copilot-custom-prompts/`. You can change this in **Settings → Copilot → Basic → Custom prompts folder**.
|
||||
|
||||
---
|
||||
|
||||
## Quick Command
|
||||
|
||||
**Quick Command** opens a modal where you can run a one-off AI prompt on your selected text without creating a permanent command.
|
||||
|
||||
- **Trigger**: Command palette → **Trigger quick command**
|
||||
- **Assign a hotkey**: Settings → Hotkeys → search "Trigger quick command"
|
||||
- **Behavior**: Opens a prompt input, lets you choose a model and whether to include the note context, then runs the prompt on your selection
|
||||
|
||||
---
|
||||
|
||||
## Quick Ask
|
||||
|
||||
**Quick Ask** is a floating inline panel that appears at the cursor position in your editor. It's designed for quick, in-context AI queries while you're writing.
|
||||
|
||||
- **Trigger**: Command palette → **Quick Ask** (or assign a hotkey, recommended: `Ctrl/Cmd+K`)
|
||||
- **Not available in Source Mode** — Works in Live Preview and Reading view
|
||||
- **How it works**: A small input appears right where your cursor is. Type your question, press Enter, and the response appears inline.
|
||||
|
||||
Quick Ask is great for things like "rephrase this sentence," "what does this term mean?", or "suggest three alternatives."
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Chat Interface](chat-interface.md) — Using slash commands in chat
|
||||
- [Context and Mentions](context-and-mentions.md) — How context is passed to commands
|
||||
- [Agent Mode and Tools](agent-mode-and-tools.md) — More powerful note editing with @composer
|
||||
148
docs/getting-started.md
Normal file
|
|
@ -0,0 +1,148 @@
|
|||
# Getting Started with Copilot for Obsidian
|
||||
|
||||
Copilot for Obsidian is an AI-powered plugin that brings large language models (LLMs) directly into your note-taking workflow. You can chat with AI, ask questions about your vault, run custom commands, search the web, and even have the AI edit your notes — all without leaving Obsidian.
|
||||
|
||||
## What Can Copilot Do?
|
||||
|
||||
- **Chat**: Have a conversation with an AI assistant
|
||||
- **Vault Q&A**: Ask questions and get answers grounded in your own notes
|
||||
- **Note editing**: Ask the AI to write or update your notes for you
|
||||
- **Semantic search**: Find notes by meaning, not just keywords
|
||||
- **Custom commands**: Run AI-powered prompts on selected text
|
||||
- **Web search**: Fetch and summarize information from the internet
|
||||
- **Memory**: Have the AI remember facts about you across conversations
|
||||
|
||||
Copilot supports 16+ AI providers including OpenAI, Anthropic, Google Gemini, Ollama (local), and more.
|
||||
|
||||
---
|
||||
|
||||
## Installation
|
||||
|
||||
1. Open **Obsidian Settings** → **Community plugins**
|
||||
2. Turn off **Safe mode** if prompted
|
||||
3. Click **Browse** and search for **Copilot**
|
||||
4. Click **Install**, then **Enable**
|
||||
|
||||
Copilot is now installed. A robot icon will appear in the left sidebar ribbon.
|
||||
|
||||
---
|
||||
|
||||
## First-Time Setup
|
||||
|
||||
### Step 1: Open Plugin Settings
|
||||
|
||||
Go to **Settings** → **Copilot** (scroll down to the Community Plugins section).
|
||||
|
||||
### Step 2: Add an API Key
|
||||
|
||||
On the **Basic** tab, click **Set Keys** to open the API key dialog. Enter the key for your chosen provider:
|
||||
|
||||
| Provider | Where to get a key |
|
||||
|---|---|
|
||||
| OpenRouter (default) | https://openrouter.ai/keys |
|
||||
| OpenAI | https://platform.openai.com/api-keys |
|
||||
| Anthropic | https://console.anthropic.com/settings/keys |
|
||||
| Google Gemini | https://makersuite.google.com/app/apikey |
|
||||
|
||||
The default model is **OpenRouter Gemini 2.5 Flash**, which requires an OpenRouter API key. If you'd prefer a different provider, set up that key first, then change the default model.
|
||||
|
||||
### Step 3: Choose a Default Model
|
||||
|
||||
Still on the **Basic** tab, use the **Default Chat Model** dropdown to select the model you want to use. Any model whose provider has an API key configured will be available.
|
||||
|
||||
### Step 4: Choose a Chat Mode
|
||||
|
||||
Use the **Default Mode** dropdown to set which mode opens by default:
|
||||
|
||||
- **Chat** — General conversation, good for most tasks
|
||||
- **Vault QA** — Ask questions answered from your notes
|
||||
- **Copilot Plus** — Advanced mode with autonomous agent and tools (requires Copilot Plus license)
|
||||
- **Projects** — Focused workspaces (alpha feature)
|
||||
|
||||
Most users should start with **Chat** mode.
|
||||
|
||||
---
|
||||
|
||||
## Opening the Chat Panel
|
||||
|
||||
You can open Copilot in several ways:
|
||||
|
||||
- Click the **robot icon** in the left ribbon (sidebar)
|
||||
- Use the command palette: `Ctrl/Cmd+P` → **Open Copilot Chat Window**
|
||||
- Use the hotkey `Ctrl/Cmd+P` → **Toggle Copilot Chat Window** to show/hide it
|
||||
|
||||
### Sidebar vs. Editor Tab
|
||||
|
||||
By default, Copilot opens as a **view** (sidebar panel). You can change this in Settings → Copilot → Basic → **Open chat in**:
|
||||
- **View** — Opens in the sidebar, stays visible as you work
|
||||
- **Editor** — Opens as an editor tab, giving it more screen space
|
||||
|
||||
---
|
||||
|
||||
## Your First Conversation
|
||||
|
||||
1. Open the chat panel
|
||||
2. Type your message in the input box at the bottom
|
||||
3. Press **Enter** (or **Shift+Enter** if you changed the send shortcut) to send
|
||||
4. Watch the AI's response stream in real time
|
||||
5. Continue the conversation naturally
|
||||
|
||||
The AI will automatically include your currently open note as context, so you can say things like "summarize this note" or "what are the action items in this note?"
|
||||
|
||||
---
|
||||
|
||||
## Keyboard Shortcuts
|
||||
|
||||
These are the default shortcuts. You can customize them in **Obsidian Settings** → **Hotkeys** → search for "Copilot".
|
||||
|
||||
| Action | Default Shortcut |
|
||||
|---|---|
|
||||
| Open Copilot Chat Window | *(unbound — assign in Hotkeys)* |
|
||||
| Toggle Copilot Chat Window | *(unbound — assign in Hotkeys)* |
|
||||
| New Copilot Chat | *(unbound — assign in Hotkeys)* |
|
||||
| Quick Ask (floating input) | *(unbound — assign in Hotkeys)* |
|
||||
| Trigger Quick Command | *(unbound — assign in Hotkeys)* |
|
||||
| Add selection to chat context | *(unbound — assign in Hotkeys)* |
|
||||
|
||||
### Send Shortcut
|
||||
|
||||
By default, **Enter** sends a message and **Shift+Enter** adds a new line. You can swap this in Settings → Copilot → Basic → **Default Send Shortcut**.
|
||||
|
||||
---
|
||||
|
||||
---
|
||||
|
||||
## Glossary
|
||||
|
||||
**LLM (Large Language Model)**
|
||||
The AI "brain" behind Copilot — a model trained on vast text to understand and generate human language, powering chat, summarization, and writing assistance.
|
||||
|
||||
**API (Application Programming Interface)**
|
||||
A way for Copilot to communicate with external AI services. You provide an API key, which is like a password that lets Copilot use a provider's AI models on your behalf. Note: an OpenAI API key is *different* from a ChatGPT Plus subscription — you don't need ChatGPT Plus to use Copilot.
|
||||
|
||||
**API Key**
|
||||
A secret token from an AI provider that authorizes Copilot to make requests. Most providers require you to have a billing account with a positive balance.
|
||||
|
||||
**Token**
|
||||
A small unit of text (roughly ¾ of a word) that AI models process. Tokens measure how much text the AI can handle at once and relate to usage costs.
|
||||
|
||||
**Context Window**
|
||||
The amount of text the AI can consider at one time when generating a response. A larger context window means the AI can use more of your notes or conversation history.
|
||||
|
||||
**Embeddings**
|
||||
A method of converting text into numbers that capture meaning. Embeddings let the AI find notes that are conceptually related, even if they don't share exact words.
|
||||
|
||||
**RAG (Retrieval-Augmented Generation)**
|
||||
A technique that enhances AI responses by first searching for relevant notes, then generating an answer based on both your query and the retrieved content. This is how Vault QA works.
|
||||
|
||||
**Vector Store / Index**
|
||||
A database that stores your notes as mathematical vectors (embeddings) so they can be searched by meaning. Think of it as a smart index that understands the context of your notes, not just their keywords.
|
||||
|
||||
---
|
||||
|
||||
## Next Steps
|
||||
|
||||
- [Chat Interface](chat-interface.md) — Learn about modes, history, and settings
|
||||
- [LLM Providers](llm-providers.md) — Set up your preferred AI provider
|
||||
- [Context and Mentions](context-and-mentions.md) — Control what context the AI sees
|
||||
- [Vault Search and Indexing](vault-search-and-indexing.md) — Set up semantic search over your notes
|
||||
29
docs/index.md
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
# Copilot for Obsidian — Documentation
|
||||
|
||||
Welcome to the official documentation for **Copilot for Obsidian**, an AI-powered assistant plugin that brings the power of large language models directly into your note-taking workflow.
|
||||
|
||||
## Table of Contents
|
||||
|
||||
| Document | What it covers |
|
||||
|---|---|
|
||||
| [Getting Started](getting-started.md) | Installation, first-time setup, opening the chat panel, keyboard shortcuts |
|
||||
| [Chat Interface](chat-interface.md) | Chat modes, sending messages, history, settings, auto-compact |
|
||||
| [LLM Providers](llm-providers.md) | All 16+ supported providers and how to set them up |
|
||||
| [Models and Parameters](models-and-parameters.md) | Chat models, embedding models, temperature, max tokens, and other parameters |
|
||||
| [Context and Mentions](context-and-mentions.md) | Active note context, @-mentions, URLs, tags, and the web viewer |
|
||||
| [Custom Commands](custom-commands.md) | Creating and using preset prompts, template variables, Quick Command, Quick Ask |
|
||||
| [Vault Search and Indexing](vault-search-and-indexing.md) | Lexical search, semantic search, index management, exclusions |
|
||||
| [Agent Mode and Tools](agent-mode-and-tools.md) | Autonomous agent, all 13 tools, file editing, web search |
|
||||
| [Projects](projects.md) | Focused workspaces with isolated context, model, and chat history |
|
||||
| [System Prompts](system-prompts.md) | Customizing AI behavior with built-in and custom system prompts |
|
||||
| [Copilot Plus and Self-Host](copilot-plus-and-self-host.md) | Copilot Plus features, memory system, self-host mode, Miyo |
|
||||
| [Troubleshooting and FAQ](troubleshooting-and-faq.md) | Common errors, provider-specific issues, performance, FAQ |
|
||||
|
||||
## Quick Start
|
||||
|
||||
1. Install Copilot from Obsidian Community Plugins
|
||||
2. Add an API key in Settings → Copilot → Basic → API Keys
|
||||
3. Open the chat panel with the robot icon in the left ribbon
|
||||
4. Start chatting!
|
||||
|
||||
For a full walkthrough, see [Getting Started](getting-started.md).
|
||||
189
docs/llm-providers.md
Normal file
|
|
@ -0,0 +1,189 @@
|
|||
# LLM Providers
|
||||
|
||||
Copilot includes 16 built-in AI providers, and you can add an unlimited number of additional models as long as they are OpenAI-compatible. You can use cloud-based services that require API keys, or run models locally on your own machine. This guide explains how to set up each provider.
|
||||
|
||||
---
|
||||
|
||||
## How to Set API Keys
|
||||
|
||||
1. Go to **Settings → Copilot → Basic**
|
||||
2. Click **Set Keys** to open the API key dialog
|
||||
3. Enter your key for the provider you want to use
|
||||
4. Click Save
|
||||
|
||||
You can configure multiple providers simultaneously and switch between them by changing the default model.
|
||||
|
||||
---
|
||||
|
||||
## Cloud Providers
|
||||
|
||||
### OpenRouter (Default)
|
||||
|
||||
OpenRouter is a gateway that provides access to hundreds of models from many providers through a single API key.
|
||||
|
||||
- **Get a key**: https://openrouter.ai/keys
|
||||
- **Default model**: OpenRouter Gemini 2.5 Flash
|
||||
- **Why use it**: One key, many models. Good starting point.
|
||||
- **Setting key**: `openRouterAiApiKey`
|
||||
|
||||
### OpenAI
|
||||
|
||||
Direct access to GPT-4.1, GPT-5, and other OpenAI models.
|
||||
|
||||
- **Get a key**: https://platform.openai.com/api-keys
|
||||
- **Models include**: GPT-5.4, GPT-5 mini, GPT-5 nano, GPT-4.1, GPT-4.1 mini, GPT-4.1 nano, o4-mini (reasoning)
|
||||
- **Setting key**: `openAIApiKey`
|
||||
|
||||
### Anthropic
|
||||
|
||||
Access to Claude models (Opus, Sonnet, etc.).
|
||||
|
||||
- **Get a key**: https://console.anthropic.com/settings/keys
|
||||
- **Models include**: claude-opus-4-6, claude-sonnet-4-5
|
||||
- **Setting key**: `anthropicApiKey`
|
||||
|
||||
### Google Gemini
|
||||
|
||||
Access to Google's Gemini family of models.
|
||||
|
||||
- **Get a key**: https://makersuite.google.com/app/apikey
|
||||
- **Models include**: gemini-2.5-pro, gemini-2.5-flash, gemini-3.5-flash, gemini-3.1-pro-preview
|
||||
- **Setting key**: `googleApiKey`
|
||||
|
||||
### XAI / Grok
|
||||
|
||||
Access to Grok models from xAI.
|
||||
|
||||
- **Get a key**: https://console.x.ai
|
||||
- **Models include**: grok-4-1-fast
|
||||
- **Setting key**: `xaiApiKey`
|
||||
|
||||
### Groq
|
||||
|
||||
Groq provides very fast inference for open-source models.
|
||||
|
||||
- **Get a key**: https://console.groq.com/keys
|
||||
- **Models include**: llama3-8b-8192 (and others)
|
||||
- **Setting key**: `groqApiKey`
|
||||
|
||||
### Mistral
|
||||
|
||||
Access to Mistral AI's models.
|
||||
|
||||
- **Get a key**: https://console.mistral.ai/api-keys
|
||||
- **Models include**: mistral-tiny-latest (and others)
|
||||
- **Setting key**: `mistralApiKey`
|
||||
|
||||
### DeepSeek
|
||||
|
||||
Access to DeepSeek's chat and reasoning models.
|
||||
|
||||
- **Get a key**: https://platform.deepseek.com/api-keys
|
||||
- **Models include**: deepseek-chat, deepseek-reasoner
|
||||
- **Setting key**: `deepseekApiKey`
|
||||
|
||||
### Cohere
|
||||
|
||||
Access to Cohere's Command models.
|
||||
|
||||
- **Get a key**: https://dashboard.cohere.ai/api-keys
|
||||
- **Models include**: command-r
|
||||
- **Setting key**: `cohereApiKey`
|
||||
|
||||
### SiliconFlow
|
||||
|
||||
A Chinese AI cloud platform with access to DeepSeek and Qwen models.
|
||||
|
||||
- **Get a key**: https://cloud.siliconflow.com/me/account/ak
|
||||
- **Models include**: DeepSeek-V3, DeepSeek-R1 (via SiliconFlow)
|
||||
- **Setting key**: `siliconflowApiKey`
|
||||
|
||||
### Azure OpenAI
|
||||
|
||||
Access to OpenAI models deployed on Microsoft Azure. Requires four fields to be configured:
|
||||
|
||||
| Setting | Description |
|
||||
| --------------- | -------------------------- |
|
||||
| API Key | Your Azure OpenAI key |
|
||||
| Instance Name | Your Azure resource name |
|
||||
| Deployment Name | Your model deployment name |
|
||||
| API Version | e.g., `2024-02-01` |
|
||||
|
||||
- **Note**: Unlike other providers, Azure OpenAI uses your own Azure deployment
|
||||
- **Embedding**: Can also use Azure for embeddings (separate deployment name required)
|
||||
|
||||
### Amazon Bedrock
|
||||
|
||||
Access to models hosted on AWS Bedrock.
|
||||
|
||||
- **Get credentials**: https://console.aws.amazon.com/iam/home#/security_credentials
|
||||
- **Required fields**: Access Key ID (API key), Region
|
||||
- **Setting key**: `amazonBedrockApiKey`
|
||||
|
||||
**Important**: Always use cross-region inference profile IDs, not bare model IDs. For example:
|
||||
|
||||
- Use: `us.anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
- Not: `anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
|
||||
Cross-region profiles (with the `us.`, `eu.`, `apac.`, or `global.` prefix) are more reliable and available across regions.
|
||||
|
||||
### GitHub Copilot
|
||||
|
||||
Use your existing GitHub Copilot subscription to access AI models.
|
||||
|
||||
- **OAuth flow**: Click **Connect GitHub Copilot** in the API key dialog
|
||||
- **No separate API key needed** — authenticates via GitHub OAuth
|
||||
- **Requires**: Active GitHub Copilot subscription
|
||||
|
||||
---
|
||||
|
||||
## Local Model Providers
|
||||
|
||||
Local providers run models on your own computer. No API key or internet connection needed once set up.
|
||||
|
||||
### Ollama
|
||||
|
||||
Runs open-source models locally on your machine.
|
||||
|
||||
- **Default port**: 11434
|
||||
- **URL**: `http://localhost:11434/v1/`
|
||||
- **Setup**: Install Ollama (ollama.ai), pull a model, then add it in Copilot's Model settings
|
||||
- **No API key required**
|
||||
|
||||
### LM Studio
|
||||
|
||||
A desktop app for running local models with a GUI.
|
||||
|
||||
- **Default port**: 1234
|
||||
- **URL**: `http://localhost:1234/v1`
|
||||
- **Setup**: Install LM Studio, load a model, go to the Developer tab, **enable CORS** (required), click "Start Server", then add the model in Copilot
|
||||
- **No API key required**
|
||||
|
||||
### 3rd Party (OpenAI-Format)
|
||||
|
||||
For any API that follows the OpenAI API format. Useful for custom deployments, proxies, or other local inference servers (vLLM, LiteLLM, etc.).
|
||||
|
||||
- **Requires**: Base URL and optionally an API key
|
||||
- **Use when**: Your provider isn't in the list but speaks OpenAI-format
|
||||
|
||||
> **CORS Warning**: Some third-party providers (e.g., Perplexity) don't support CORS, which causes Copilot to fail with a CORS error. When adding a custom model for such a provider, enable the **CORS** toggle in the custom model form. Note: streaming is not available in CORS mode.
|
||||
|
||||
---
|
||||
|
||||
## Provider-Specific Gotchas
|
||||
|
||||
| Provider | Common Issue | Fix |
|
||||
| -------------- | ----------------------------------- | -------------------------------------------------------------------------------------- |
|
||||
| Azure OpenAI | Missing one of four required fields | Check all four settings: key, instance name, deployment name, API version |
|
||||
| Amazon Bedrock | Rate limit or model not found | Use cross-region inference profile IDs with `us.`, `eu.`, `apac.`, or `global.` prefix |
|
||||
| GitHub Copilot | Token expired | Re-authenticate via the OAuth button in API key dialog |
|
||||
| Ollama | Connection refused | Make sure Ollama is running (`ollama serve`) and the port is correct |
|
||||
| Google Gemini | Quota exceeded | Use a different model or check your quota at console.cloud.google.com |
|
||||
| DeepSeek | Streaming errors | Try disabling streaming in the per-session settings if you encounter issues |
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Models and Parameters](models-and-parameters.md) — Enable, disable, and configure models
|
||||
- [Getting Started](getting-started.md) — First-time setup
|
||||
471
docs/miyo-api.md
Normal file
|
|
@ -0,0 +1,471 @@
|
|||
# Miyo Node Service API
|
||||
|
||||
Base URL: `http://127.0.0.1:8742`
|
||||
|
||||
All request and response bodies are JSON. Errors always return `{ "detail": "<message>" }`.
|
||||
|
||||
---
|
||||
|
||||
## Health
|
||||
|
||||
### `GET /v0/health`
|
||||
|
||||
Returns service and sidecar status.
|
||||
|
||||
**Response 200**
|
||||
|
||||
```json
|
||||
{
|
||||
"status": "ok | degraded",
|
||||
"service": "running",
|
||||
"qdrant": "connected | ...",
|
||||
"llama_server": "running | ...",
|
||||
"model_download_progress": 0.75,
|
||||
"embedding_model": "nomic-embed-text-v1.5",
|
||||
"batch_size_preset": "default",
|
||||
"gpu_variant": "metal | null",
|
||||
"indexed_files": 1234
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Search
|
||||
|
||||
### `POST /v0/search`
|
||||
|
||||
Hybrid semantic + keyword search (dense + BM25, fused via RRF).
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{
|
||||
"query": "string (required)",
|
||||
"folder_path": "string | null — restrict to this folder",
|
||||
"path": "string | null — substring filter on file path (case-insensitive)",
|
||||
"limit": 10,
|
||||
"filters": [
|
||||
/* MetadataFilter[], see below */
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
**Response 200**
|
||||
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
/* SearchResult[] */
|
||||
],
|
||||
"query": "string",
|
||||
"count": 5,
|
||||
"execution_time_ms": 42.0
|
||||
}
|
||||
```
|
||||
|
||||
**Errors:** 400 (missing query), 503 (llama-server or Qdrant unavailable)
|
||||
|
||||
---
|
||||
|
||||
### `POST /v0/search/related`
|
||||
|
||||
Find files related to a given file using vector similarity.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{
|
||||
"file_path": "string (required) — absolute path",
|
||||
"folder_path": "string | null",
|
||||
"limit": 10,
|
||||
"filters": [
|
||||
/* MetadataFilter[] */
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
**Response 200**
|
||||
|
||||
```json
|
||||
{
|
||||
"results": [{ "path": "string", "score": 0.95 }],
|
||||
"file_path": "string",
|
||||
"count": 5,
|
||||
"execution_time_ms": 12.0
|
||||
}
|
||||
```
|
||||
|
||||
**Errors:** 400, 404 (no indexed chunks for the file), 503
|
||||
|
||||
---
|
||||
|
||||
## Folders
|
||||
|
||||
### `GET /v0/folder`
|
||||
|
||||
- With `?path=<folder_path>`: returns a single `FolderEntry` (404 if not registered)
|
||||
- Without `path`: returns `{ "folders": [ FolderEntry[] ] }`
|
||||
|
||||
---
|
||||
|
||||
### `POST /v0/folder`
|
||||
|
||||
Register a folder for indexing. Starts watching and scanning immediately.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{
|
||||
"path": "string (required) — absolute path",
|
||||
"include_patterns": ["**/*.md"],
|
||||
"exclude_patterns": ["**/node_modules/**"],
|
||||
"recursive": true
|
||||
}
|
||||
```
|
||||
|
||||
**Response 201** — `FolderEntry`
|
||||
|
||||
**Errors:** 400 (invalid), 409 (already registered)
|
||||
|
||||
---
|
||||
|
||||
### `PATCH /v0/folder`
|
||||
|
||||
Update folder configuration. Only provided fields are changed.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{
|
||||
"path": "string (required)",
|
||||
"include_patterns": ["**/*.md"],
|
||||
"exclude_patterns": ["**/node_modules/**"],
|
||||
"recursive": false
|
||||
}
|
||||
```
|
||||
|
||||
**Response 200** — updated `FolderEntry`
|
||||
|
||||
**Errors:** 400, 404
|
||||
|
||||
---
|
||||
|
||||
### `DELETE /v0/folder`
|
||||
|
||||
Unregister a folder and remove all its indexed data.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{ "path": "string (required)" }
|
||||
```
|
||||
|
||||
**Response 200** — deletion summary object
|
||||
|
||||
**Errors:** 400, 404
|
||||
|
||||
---
|
||||
|
||||
### `POST /v0/folder/pause`
|
||||
|
||||
Stop file watching for a folder without removing it.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{ "path": "string (required)" }
|
||||
```
|
||||
|
||||
**Response 200**
|
||||
|
||||
```json
|
||||
{ "status": "paused", "path": "string" }
|
||||
```
|
||||
|
||||
**Errors:** 400, 404
|
||||
|
||||
---
|
||||
|
||||
### `POST /v0/folder/resume`
|
||||
|
||||
Resume file watching and trigger a rescan.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{ "path": "string (required)" }
|
||||
```
|
||||
|
||||
**Response 202**
|
||||
|
||||
```json
|
||||
{ "status": "scanning", "path": "string" }
|
||||
```
|
||||
|
||||
**Errors:** 400, 404
|
||||
|
||||
---
|
||||
|
||||
### `POST /v0/scan`
|
||||
|
||||
Manually trigger a rescan of a registered folder.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{
|
||||
"path": "string (required)",
|
||||
"force": false
|
||||
}
|
||||
```
|
||||
|
||||
`force: true` re-indexes all files even if unchanged.
|
||||
|
||||
**Response 202**
|
||||
|
||||
```json
|
||||
{ "status": "started", "path": "string" }
|
||||
```
|
||||
|
||||
**Errors:** 400, 404
|
||||
|
||||
---
|
||||
|
||||
## Files & Documents
|
||||
|
||||
### `GET /v0/folder/files`
|
||||
|
||||
List indexed files with optional filtering and pagination.
|
||||
|
||||
**Query parameters**
|
||||
|
||||
| Param | Type | Description |
|
||||
| -------------- | --------------------------------- | ------------------------------- |
|
||||
| `folder_path` | string | Filter by folder |
|
||||
| `title` | string | Substring match on title |
|
||||
| `file_path` | string | Exact file path match |
|
||||
| `mtime_after` | number | Unix timestamp lower bound |
|
||||
| `mtime_before` | number | Unix timestamp upper bound |
|
||||
| `offset` | integer (default 0) | Pagination offset |
|
||||
| `limit` | integer | Max results (omit for no limit) |
|
||||
| `order_by` | `mtime` \| `updated_at` (default) | Sort order |
|
||||
|
||||
**Response 200**
|
||||
|
||||
```json
|
||||
{
|
||||
"files": [
|
||||
/* FileEntry[] */
|
||||
],
|
||||
"total": 99
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### `GET /v0/folder/documents`
|
||||
|
||||
Fetch all indexed chunks for a specific file, sorted by chunk index.
|
||||
|
||||
**Query parameters**
|
||||
|
||||
| Param | Required | Description |
|
||||
| ------------- | -------- | -------------------------- |
|
||||
| `path` | yes | Absolute file path |
|
||||
| `folder_path` | no | Scope to a specific folder |
|
||||
|
||||
**Response 200**
|
||||
|
||||
```json
|
||||
{
|
||||
"documents": [
|
||||
/* DocumentChunk[] */
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
**Errors:** 400, 503
|
||||
|
||||
---
|
||||
|
||||
## Utilities
|
||||
|
||||
### `POST /v0/parse-doc`
|
||||
|
||||
Parse a file and return its extracted text content.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{ "path": "string (required) — absolute file path" }
|
||||
```
|
||||
|
||||
**Response 200** — parsed content object (shape varies by file type)
|
||||
|
||||
**Errors:**
|
||||
|
||||
| Code | Meaning |
|
||||
| ---- | --------------------- |
|
||||
| 400 | Invalid input |
|
||||
| 403 | File not readable |
|
||||
| 404 | File not found |
|
||||
| 415 | Unsupported file type |
|
||||
| 422 | Parse failed |
|
||||
| 500 | Internal error |
|
||||
|
||||
---
|
||||
|
||||
### `POST /v0/rebuild-metadata`
|
||||
|
||||
Rebuild the manifest by re-syncing metadata from Qdrant. Use when manifest is out of sync.
|
||||
|
||||
**Response 200** — `{ "elapsed_ms": 123, ...stats }`
|
||||
|
||||
**Errors:** 409 (rebuild already in progress), 503
|
||||
|
||||
---
|
||||
|
||||
### `POST /v0/llama-server/restart`
|
||||
|
||||
Restart the llama-server sidecar, optionally changing the batch size preset.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{ "batch_size_preset": "default" }
|
||||
```
|
||||
|
||||
**Response 200**
|
||||
|
||||
```json
|
||||
{
|
||||
"restarted": true,
|
||||
"batch_size_preset": "default",
|
||||
"status": "running"
|
||||
}
|
||||
```
|
||||
|
||||
**Errors:** 400
|
||||
|
||||
---
|
||||
|
||||
### `POST /v1/embeddings`
|
||||
|
||||
Generate embeddings. OpenAI-compatible interface, proxied to llama-server.
|
||||
|
||||
**Request body**
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "nomic-embed-text-v1.5",
|
||||
"input": "string or string[]"
|
||||
}
|
||||
```
|
||||
|
||||
`model` is optional but must match the configured embedding model if provided.
|
||||
|
||||
**Response 200** — standard OpenAI embeddings response
|
||||
|
||||
**Errors:** 400, 503
|
||||
|
||||
---
|
||||
|
||||
## Schemas
|
||||
|
||||
### MetadataFilter
|
||||
|
||||
Range filter on a metadata field.
|
||||
|
||||
```json
|
||||
{
|
||||
"field": "mtime",
|
||||
"gt": 1700000000,
|
||||
"gte": 1700000000,
|
||||
"lt": 1800000000,
|
||||
"lte": 1800000000
|
||||
}
|
||||
```
|
||||
|
||||
- `field` can be `mtime`, `ctime`, or any metadata key
|
||||
- Bare field names (not `mtime`/`ctime` and not prefixed with `metadata.`) are automatically prefixed with `metadata.`
|
||||
- At least one of `gt`, `gte`, `lt`, `lte` must be present
|
||||
|
||||
---
|
||||
|
||||
### SearchResult
|
||||
|
||||
```json
|
||||
{
|
||||
"path": "string",
|
||||
"score": 0.95,
|
||||
"title": "string | null",
|
||||
"mtime": 1700000000,
|
||||
"ctime": 1700000000,
|
||||
"file_name": "string | null",
|
||||
"chunk_index": 0,
|
||||
"total_chunks": 5,
|
||||
"chunk_text": "string | null",
|
||||
"metadata": {},
|
||||
"embedding_model": "string | null",
|
||||
"tags": ["string"],
|
||||
"extension": ".md",
|
||||
"created_at": "string | null",
|
||||
"nchars": 1024,
|
||||
"folder_path": "string | null"
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### FileEntry
|
||||
|
||||
```json
|
||||
{
|
||||
"path": "string",
|
||||
"title": "string | null",
|
||||
"mtime": 1700000000,
|
||||
"updated_at": "ISO8601 string",
|
||||
"folder_path": "string | null",
|
||||
"total_chunks": 5
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### DocumentChunk
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "string",
|
||||
"path": "string | null",
|
||||
"title": "string | null",
|
||||
"chunk_index": 0,
|
||||
"chunk_text": "string | null",
|
||||
"metadata": {},
|
||||
"embedding_model": "string | null",
|
||||
"ctime": 1700000000,
|
||||
"mtime": 1700000000,
|
||||
"tags": ["string"],
|
||||
"extension": ".md",
|
||||
"created_at": "ISO8601 string | null",
|
||||
"nchars": 1024,
|
||||
"folder_path": "string | null"
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### FolderEntry
|
||||
|
||||
Shape varies — includes at minimum:
|
||||
|
||||
```json
|
||||
{
|
||||
"path": "string",
|
||||
"include_patterns": ["**/*.md"],
|
||||
"exclude_patterns": [],
|
||||
"recursive": true
|
||||
}
|
||||
```
|
||||
|
||||
Plus live stats fields populated by the folder manager.
|
||||
178
docs/models-and-parameters.md
Normal file
|
|
@ -0,0 +1,178 @@
|
|||
# Models and Parameters
|
||||
|
||||
This guide explains how to manage chat models, embedding models, and the parameters that control how the AI behaves.
|
||||
|
||||
---
|
||||
|
||||
## Chat Models
|
||||
|
||||
### Built-In Models
|
||||
|
||||
Copilot comes with a set of built-in models across many providers. Some are always included ("core" models); others can be enabled or disabled.
|
||||
|
||||
| Model | Provider | Capabilities |
|
||||
| ----------------------------- | ------------ | ----------------------- |
|
||||
| copilot-plus-flash | Copilot Plus | Vision (Plus exclusive) |
|
||||
| google/gemini-2.5-flash | OpenRouter | Vision |
|
||||
| google/gemini-2.5-pro | OpenRouter | Vision |
|
||||
| google/gemini-3.5-flash | OpenRouter | Vision, Reasoning |
|
||||
| google/gemini-3.1-pro-preview | OpenRouter | Vision, Reasoning |
|
||||
| openai/gpt-5.4 | OpenRouter | Vision |
|
||||
| openai/gpt-5-mini | OpenRouter | Vision |
|
||||
| gpt-5.4 | OpenAI | Vision |
|
||||
| gpt-5-mini | OpenAI | Vision |
|
||||
| gpt-4.1 | OpenAI | Vision |
|
||||
| gpt-4.1-mini | OpenAI | Vision |
|
||||
| claude-opus-4-6 | Anthropic | Vision, Reasoning |
|
||||
| claude-sonnet-4-5-20250929 | Anthropic | Vision, Reasoning |
|
||||
| gemini-2.5-pro | Google | Vision |
|
||||
| gemini-2.5-flash | Google | Vision |
|
||||
| gemini-3.5-flash | Google | Vision, Reasoning |
|
||||
| grok-4-1-fast | XAI | Vision |
|
||||
| deepseek-chat | DeepSeek | — |
|
||||
| deepseek-reasoner | DeepSeek | Reasoning |
|
||||
|
||||
### Model Capability Badges
|
||||
|
||||
Models may show capability badges:
|
||||
|
||||
- **Reasoning** — Extended internal thinking before responding; better for complex tasks
|
||||
- **Vision** — Can process images (e.g., screenshots, diagrams embedded in notes)
|
||||
- **Web Search** — Can access the internet directly (model-native feature)
|
||||
|
||||
### Managing Models
|
||||
|
||||
Go to **Settings → Copilot → Model** to see the full model list.
|
||||
|
||||
- **Enable/disable** — Toggle individual models on or off to control what appears in the model selector
|
||||
- **Reorder** — Drag models to change their order in the dropdown
|
||||
- **Delete** — Remove custom models you've added
|
||||
|
||||
### Adding Custom Models
|
||||
|
||||
If your provider offers a model that isn't in the built-in list, you can add it manually:
|
||||
|
||||
1. Go to **Settings → Copilot → Model**
|
||||
2. Click **Add Model**
|
||||
3. Enter the model name exactly as the provider expects it (e.g., `gpt-4-turbo-preview`)
|
||||
4. Select the provider
|
||||
5. Optionally set a custom base URL (useful for proxies or alternate endpoints)
|
||||
6. Save
|
||||
|
||||
### Importing Models from Provider
|
||||
|
||||
You can automatically import the full list of available models from a provider:
|
||||
|
||||
1. Go to **Settings → Copilot → Model**
|
||||
2. Find the **Import models** button for your provider
|
||||
3. Copilot will fetch the provider's model list and add new ones
|
||||
|
||||
---
|
||||
|
||||
## Embedding Models
|
||||
|
||||
Embedding models convert text into numerical vectors, which powers semantic (meaning-based) search in Vault QA and the "Relevant Notes" feature.
|
||||
|
||||
### Built-In Embedding Models
|
||||
|
||||
| Model | Provider |
|
||||
| ----------------------------- | --------------------------------- |
|
||||
| copilot-plus-small | Copilot Plus (Plus exclusive) |
|
||||
| copilot-plus-large | Copilot Plus (Believer exclusive) |
|
||||
| copilot-plus-multilingual | Copilot Plus (Plus exclusive) |
|
||||
| openai/text-embedding-3-small | OpenRouter |
|
||||
| text-embedding-3-small | OpenAI |
|
||||
| text-embedding-3-large | OpenAI |
|
||||
| embed-multilingual-light-v3.0 | Cohere |
|
||||
| text-embedding-004 | Google |
|
||||
| gemini-embedding-001 | Google |
|
||||
| Qwen3-Embedding-0.6B | SiliconFlow |
|
||||
|
||||
### Selecting an Embedding Model
|
||||
|
||||
Go to **Settings → Copilot → QA** → **Embedding Model**.
|
||||
|
||||
If you change embedding models, you must rebuild the vault index because the old vectors are incompatible with the new model. Copilot will prompt you to confirm before rebuilding.
|
||||
|
||||
### What Embeddings Affect
|
||||
|
||||
- **Vault QA mode** — Uses embeddings to find relevant notes by meaning
|
||||
- **Semantic Search** — The "Enable Semantic Search" toggle in QA settings
|
||||
- **Relevant Notes** — Shows semantically similar notes in the sidebar
|
||||
|
||||
---
|
||||
|
||||
## Model Parameters
|
||||
|
||||
These settings control how the AI responds. Global defaults live in Settings → Copilot → Model. You can override them per-session using the gear icon in the chat panel.
|
||||
|
||||
### Temperature
|
||||
|
||||
Controls how random or creative the responses are.
|
||||
|
||||
- **Range**: 0.0–1.0
|
||||
- **Default**: 0.1
|
||||
- **Low (0.0–0.2)**: Precise, factual, deterministic
|
||||
- **Medium (0.4–0.6)**: Balanced
|
||||
- **High (0.8–1.0)**: Creative, varied, less predictable
|
||||
|
||||
### Max Tokens
|
||||
|
||||
Maximum number of tokens in the AI's response. A **token** is roughly ¾ of a word (so 1,000 tokens ≈ 750 words).
|
||||
|
||||
- **Default**: 6,000
|
||||
- Higher values allow longer responses but cost more
|
||||
|
||||
### Conversation Turns in Context
|
||||
|
||||
How many past conversation turns to include in each request. More turns = more context but larger requests.
|
||||
|
||||
- **Default**: 15 turns
|
||||
- Reduce this if you hit context limits or want to lower costs
|
||||
|
||||
### Auto-Compact Threshold
|
||||
|
||||
When the conversation reaches this many tokens, older messages are automatically summarized.
|
||||
|
||||
- **Default**: 128,000 tokens
|
||||
- **Range**: 64,000–1,000,000 tokens
|
||||
- See [Chat Interface](chat-interface.md#auto-compact) for details
|
||||
|
||||
### Reasoning Effort
|
||||
|
||||
For reasoning-capable models (like deepseek-reasoner, claude-opus-4-6), controls how much internal reasoning the model does before responding.
|
||||
|
||||
- **Options**: minimal, low, medium, high, xhigh
|
||||
- **Default**: low
|
||||
- Higher effort = better results on complex tasks, slower responses
|
||||
|
||||
### Verbosity
|
||||
|
||||
For models that support it, controls response length and detail.
|
||||
|
||||
- **Options**: low, medium, high
|
||||
- **Default**: medium
|
||||
|
||||
### Top P
|
||||
|
||||
An alternative to temperature for controlling randomness. Leave at default unless you have a specific reason to change it.
|
||||
|
||||
### Frequency Penalty
|
||||
|
||||
Reduces the likelihood of the model repeating itself.
|
||||
|
||||
---
|
||||
|
||||
## Default Model Selection
|
||||
|
||||
Your **default model** is the one Copilot uses when you open a new chat. Set it in:
|
||||
**Settings → Copilot → Basic → Default Chat Model**
|
||||
|
||||
The default is **OpenRouter Gemini 2.5 Flash** (requires OpenRouter API key).
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [LLM Providers](llm-providers.md) — Set up API keys for your provider
|
||||
- [Vault Search and Indexing](vault-search-and-indexing.md) — How embedding models are used
|
||||
125
docs/projects.md
Normal file
|
|
@ -0,0 +1,125 @@
|
|||
# Projects
|
||||
|
||||
Projects are focused AI workspaces. Each project has its own model, system prompt, context sources, and completely isolated chat history. Use projects to keep separate AI conversations per client, topic, or area of work.
|
||||
|
||||
Projects support **50+ file types** beyond markdown, including PDFs, Word documents, PowerPoint, Excel, images, and more — making them ideal for analyzing large or diverse document collections.
|
||||
|
||||
> **Note**: Projects is an alpha feature. It may have rough edges and is subject to change.
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
In regular chat, all conversations share the same settings and model. Projects let you create dedicated workspaces with:
|
||||
|
||||
- **A specific context** — Specific notes, folders, URLs, or YouTube videos the AI always has access to
|
||||
- **A dedicated model** — Different projects can use different AI models
|
||||
- **A custom system prompt** — Each project can have its own instructions for the AI
|
||||
- **Isolated chat history** — Conversations in one project don't mix with conversations in another
|
||||
|
||||
**Example use cases:**
|
||||
- A "Research" project that always has your research notes as context
|
||||
- A "Client Work" project with a specific system prompt and access to client-related notes
|
||||
- A "Learning" project with YouTube video URLs for study materials
|
||||
|
||||
---
|
||||
|
||||
## Creating a Project
|
||||
|
||||
1. Open the chat panel
|
||||
2. Click the mode selector at the top of the chat
|
||||
3. Select **Projects (alpha)**
|
||||
4. Click **New Project** (or the `+` button)
|
||||
5. Fill in the project details and save
|
||||
|
||||
---
|
||||
|
||||
## Project Configuration
|
||||
|
||||
Each project has the following settings:
|
||||
|
||||
### Name
|
||||
A short name for the project. Appears in the project list.
|
||||
|
||||
### Description
|
||||
An optional description of what the project is for.
|
||||
|
||||
### Model
|
||||
Choose which AI model to use for this project. The available options depend on which models you have enabled.
|
||||
|
||||
### Model Settings
|
||||
Override the default temperature and max tokens specifically for this project.
|
||||
|
||||
### System Prompt
|
||||
Set a custom system prompt for this project. This replaces (or supplements) the global default. See [System Prompts](system-prompts.md) for details.
|
||||
|
||||
---
|
||||
|
||||
## Context Sources
|
||||
|
||||
Projects let you pre-load context that is always available in the project's chat.
|
||||
|
||||
### File Inclusions and Exclusions
|
||||
|
||||
Specify which notes or folders to include in this project's context:
|
||||
|
||||
- **Inclusions**: Only these notes/folders are available for search and context
|
||||
- **Exclusions**: These notes/folders are excluded from context
|
||||
|
||||
This scopes the AI's knowledge to just the notes relevant to your project.
|
||||
|
||||
### Web URLs
|
||||
|
||||
Add web page URLs that are fetched and included as context for every conversation in this project. Useful for documentation, reference pages, or web resources you frequently consult.
|
||||
|
||||
### YouTube URLs
|
||||
|
||||
Add YouTube video URLs whose transcripts are loaded into context for every conversation.
|
||||
|
||||
---
|
||||
|
||||
## Working in a Project
|
||||
|
||||
### Switching Projects
|
||||
|
||||
Use the project selector at the top of the chat panel to switch between projects. When you switch, the chat history clears and the new project's context loads.
|
||||
|
||||
### Isolated Chat History
|
||||
|
||||
Each project maintains its own chat history, completely separate from other projects and from regular (non-project) chat. Conversations don't bleed across projects.
|
||||
|
||||
### Context Loading
|
||||
|
||||
When you open a project, Copilot loads the configured context (notes, URLs, etc.) automatically. For large projects with many notes, this may take a moment.
|
||||
|
||||
---
|
||||
|
||||
## Project List Management
|
||||
|
||||
Go to the project selector to manage your projects:
|
||||
|
||||
- **Sort**: Projects can be sorted by most recently used or alphabetically
|
||||
- **Edit**: Click the edit icon to change a project's settings
|
||||
- **Delete**: Remove the project entry from the list (saved conversation files in your vault are not deleted)
|
||||
|
||||
Sort strategy: **Settings → Copilot → Basic → Project list sort strategy**
|
||||
|
||||
---
|
||||
|
||||
## Limitations
|
||||
|
||||
As an alpha feature, projects have some known limitations:
|
||||
|
||||
- Large context sources (many notes or large files) may slow down context loading
|
||||
- The context loading on project switch is synchronous — the AI isn't available until loading completes
|
||||
- Some features available in regular Plus mode may behave differently in projects
|
||||
- Auto-compact behavior is the same as regular chat
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Chat Interface](chat-interface.md) — Chat modes overview, new chat behavior, history
|
||||
- [System Prompts](system-prompts.md) — Custom system prompts for projects
|
||||
- [Context and Mentions](context-and-mentions.md) — How context works
|
||||
- [Copilot Plus and Self-Host](copilot-plus-and-self-host.md) — Plus features
|
||||
126
docs/system-prompts.md
Normal file
|
|
@ -0,0 +1,126 @@
|
|||
# System Prompts
|
||||
|
||||
A system prompt is a set of instructions you give the AI that shapes how it behaves in all conversations. Think of it as a persistent briefing: "You are an assistant that helps me with academic writing. Always cite sources. Respond in formal English."
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Copilot has two layers of system prompts:
|
||||
|
||||
1. **Built-in system prompt** — Always active. Defines core behaviors specific to Obsidian (how to format Obsidian links, how to handle note references, etc.)
|
||||
2. **Custom system prompt** — Optional. You can write your own instructions that are appended to the built-in prompt.
|
||||
|
||||
---
|
||||
|
||||
## Built-In System Prompt
|
||||
|
||||
The built-in prompt is always active and cannot be edited. It tells the AI:
|
||||
|
||||
- It is "Obsidian Copilot" — an AI integrated into Obsidian
|
||||
- How to format Obsidian internal links: `[[Note Title]]`
|
||||
- How to format Obsidian image links: `![[image.png]]`
|
||||
- How to format LaTeX math: use `$...$` not `\[...\]`
|
||||
- How to handle @vault and @tool mentions
|
||||
- To use `-` for bullet points (not `*`)
|
||||
- To respond in the language of the user's query
|
||||
- To treat "note" as referring to an Obsidian note
|
||||
|
||||
This prompt ensures Copilot's output is correctly formatted for Obsidian and aware of its context.
|
||||
|
||||
> **Warning**: Disabling the built-in prompt can break features like Vault QA, memory, and agent tools. Avoid disabling it unless you have a specific reason.
|
||||
|
||||
---
|
||||
|
||||
## Custom System Prompts
|
||||
|
||||
Custom system prompts let you add your own instructions on top of the built-in prompt.
|
||||
|
||||
### Where They're Stored
|
||||
|
||||
Custom system prompts are stored as markdown files in your vault, in the folder:
|
||||
```
|
||||
copilot/system-prompts/
|
||||
```
|
||||
|
||||
You can change this folder in **Settings → Copilot → Advanced → System Prompts Folder Name**.
|
||||
|
||||
### Creating a System Prompt
|
||||
|
||||
#### From Settings
|
||||
|
||||
1. Go to **Settings → Copilot → Advanced**
|
||||
2. Under **User System Prompt**, click the `+` button
|
||||
3. Enter a title for the prompt (e.g., "Academic Writing")
|
||||
4. A new markdown file is created in your system prompts folder
|
||||
5. Open the file and write your instructions
|
||||
|
||||
#### From the System Prompts Folder
|
||||
|
||||
Create any `.md` file in the `copilot/system-prompts/` folder. Its filename (without `.md`) becomes the prompt's title.
|
||||
|
||||
### Writing Good System Prompts
|
||||
|
||||
Tips for effective system prompts:
|
||||
|
||||
- **Be specific**: "Always respond in bullet points with no more than 5 bullets" is better than "be concise"
|
||||
- **Set a persona**: "You are an expert in cognitive science helping me build a Zettelkasten"
|
||||
- **Define output format**: Specify if you want headers, lists, prose, or code blocks
|
||||
- **Set language**: "Always respond in French" if you want non-English output
|
||||
- **Limit scope**: "Only answer questions related to my research notes on climate science"
|
||||
|
||||
**Example system prompt:**
|
||||
```markdown
|
||||
You are a Zettelkasten assistant helping me build a knowledge base.
|
||||
- Always connect new ideas to existing notes when possible
|
||||
- Suggest up to 3 related concepts per response
|
||||
- Format all note suggestions as [[Note Title]]
|
||||
- Keep responses concise — under 200 words
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Setting a Global Default
|
||||
|
||||
You can set one of your custom prompts as the global default — it will be used for all new chat sessions:
|
||||
|
||||
1. Go to **Settings → Copilot → Advanced**
|
||||
2. Under **Default System Prompt**, select your prompt from the dropdown
|
||||
3. Any new conversation will start with this prompt active
|
||||
|
||||
To stop using a custom default, select **None (use built-in prompt)** from the dropdown.
|
||||
|
||||
---
|
||||
|
||||
## Per-Session Override (Gear Icon)
|
||||
|
||||
You can override the system prompt for just the current conversation:
|
||||
|
||||
1. Click the **gear icon** in the chat panel toolbar
|
||||
2. Select a different system prompt (or type a one-off prompt directly)
|
||||
3. This applies to the current session only and resets when you start a new chat
|
||||
|
||||
---
|
||||
|
||||
## How Prompts Combine
|
||||
|
||||
When you have a custom prompt active:
|
||||
|
||||
1. The built-in Copilot prompt runs first
|
||||
2. Your custom prompt is appended after it
|
||||
|
||||
Both sets of instructions are active simultaneously. Your custom instructions can refine, restrict, or extend the default behavior, but they don't replace it.
|
||||
|
||||
---
|
||||
|
||||
## Per-Project System Prompts
|
||||
|
||||
Each [Project](projects.md) can have its own system prompt, independent of the global default. Configure this in the project settings under **System Prompt**.
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Chat Interface](chat-interface.md) — Per-session gear settings
|
||||
- [Projects](projects.md) — Per-project system prompts
|
||||
- [Getting Started](getting-started.md) — Initial setup
|
||||
307
docs/troubleshooting-and-faq.md
Normal file
|
|
@ -0,0 +1,307 @@
|
|||
# Troubleshooting and FAQ
|
||||
|
||||
This guide covers common errors, provider-specific issues, performance problems, and frequently asked questions.
|
||||
|
||||
---
|
||||
|
||||
## First Steps for Any Issue
|
||||
|
||||
Before diving into specific fixes, try these steps first:
|
||||
|
||||
1. **Check you're on the latest version** of Copilot in Community Plugins
|
||||
2. **Disable other plugins** temporarily to rule out conflicts
|
||||
3. **Enable Debug Mode** in Settings → Copilot → Advanced → Debug Mode
|
||||
4. **Open the developer console**: `Cmd+Option+I` on Mac, `Ctrl+Shift+I` on Windows
|
||||
|
||||
---
|
||||
|
||||
## Common Errors
|
||||
|
||||
### "API key not set" or "No API key configured"
|
||||
|
||||
**Cause**: The model you selected doesn't have a valid API key for its provider.
|
||||
|
||||
**Fix**:
|
||||
1. Go to **Settings → Copilot → Basic → Set Keys**
|
||||
2. Enter the API key for the provider your model uses
|
||||
3. If you're unsure which provider a model uses, check **Settings → Copilot → Model** — each model shows its provider
|
||||
|
||||
### Rate Limit Errors
|
||||
|
||||
**Cause**: You've sent too many requests to the API in a short time.
|
||||
|
||||
**Fix**:
|
||||
- Wait a minute and try again
|
||||
- If this happens frequently during indexing, reduce **Embedding Requests per Minute** in QA settings (try 10–20)
|
||||
- Consider upgrading your API plan with the provider
|
||||
|
||||
### Connection Errors / Timeout
|
||||
|
||||
**Cause**: Network issue, provider outage, or the request took too long.
|
||||
|
||||
**Fix**:
|
||||
- Check your internet connection
|
||||
- Try again after a few seconds
|
||||
- Check the provider's status page for outages
|
||||
- If using a local model (Ollama/LM Studio), make sure the local server is running
|
||||
|
||||
### "Copilot index does not exist"
|
||||
|
||||
**Cause**: You're trying to use Vault QA or semantic search but the vault hasn't been indexed yet.
|
||||
|
||||
**Fix**:
|
||||
1. Make sure you have an embedding model configured with a valid API key (**Settings → Copilot → QA → Embedding Model**)
|
||||
2. Run **Command palette → Index (refresh) vault**
|
||||
3. Wait for indexing to complete
|
||||
|
||||
### "RangeError: invalid string length"
|
||||
|
||||
**Cause**: Your vault is too large for a single index partition.
|
||||
|
||||
**Fix**: Increase the number of partitions in **Settings → Copilot → QA → Partitions**. A good target is keeping the first index file under ~400 MB (check the `.obsidian/` folder for `copilot-index` files and their sizes).
|
||||
|
||||
### Response Gets Cut Off
|
||||
|
||||
**Cause**: The AI's response hit the Max Tokens limit.
|
||||
|
||||
**Fix**: Increase **Max Tokens** in Settings → Copilot → Model (or the per-session gear icon). Default is 6,000 tokens.
|
||||
|
||||
### Notes Not Found in Search
|
||||
|
||||
Even after indexing, relevant notes aren't being returned? Try:
|
||||
1. Switch to **Copilot Plus** mode and use `@vault` for more powerful search
|
||||
2. Try the **multilingual embedding model** for non-English notes
|
||||
3. Review your QA inclusions/exclusions to confirm the notes aren't filtered out
|
||||
4. Run **List all indexed files** (debug command) to verify the notes are indexed
|
||||
5. Run **Force reindex vault** for a clean rebuild
|
||||
|
||||
### "Non-markdown files are only available in Copilot Plus"
|
||||
|
||||
**Cause**: You tried to use a PDF, image, or other non-markdown file as context in a free mode.
|
||||
|
||||
**Fix**: Switch to Copilot Plus mode, or convert the file to markdown manually.
|
||||
|
||||
---
|
||||
|
||||
## Provider-Specific Issues
|
||||
|
||||
### Ollama
|
||||
|
||||
**Problem**: "Connection refused" or model not responding
|
||||
|
||||
**Fix**:
|
||||
- Make sure Ollama is running: open a terminal and run `ollama serve`
|
||||
- Verify the model is downloaded: `ollama list`
|
||||
- Check that the port in Copilot settings matches (default: 11434)
|
||||
- On some systems, Ollama uses `http://127.0.0.1:11434` instead of `http://localhost:11434` — try both
|
||||
|
||||
### Azure OpenAI
|
||||
|
||||
**Problem**: Authentication errors or model not found
|
||||
|
||||
**Fix**:
|
||||
Azure OpenAI requires all four fields to be filled in correctly:
|
||||
1. API Key
|
||||
2. Instance Name (your Azure resource name, e.g., `my-azure-openai`)
|
||||
3. Deployment Name (the name you gave your model deployment)
|
||||
4. API Version (e.g., `2024-02-01`)
|
||||
|
||||
Any missing or incorrect field will cause errors.
|
||||
|
||||
### Amazon Bedrock
|
||||
|
||||
**Problem**: "Model not found" or access denied
|
||||
|
||||
**Fix**:
|
||||
- Always use **cross-region inference profile IDs**, not bare model IDs:
|
||||
- ✅ `us.anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
- ❌ `anthropic.claude-sonnet-4-5-20250929-v1:0`
|
||||
- Make sure your IAM credentials have Bedrock access permissions
|
||||
- Confirm the model is available in your region
|
||||
|
||||
### GitHub Copilot
|
||||
|
||||
**Problem**: "Token expired" or authentication fails
|
||||
|
||||
**Fix**:
|
||||
- Go to **Settings → Copilot → Basic → Set Keys**
|
||||
- Click **Connect GitHub Copilot** to re-authenticate via OAuth
|
||||
- Make sure your GitHub Copilot subscription is active
|
||||
|
||||
### Google Gemini
|
||||
|
||||
**Problem**: "QUOTA_EXCEEDED" or slow responses
|
||||
|
||||
**Fix**:
|
||||
- Check your quota at https://console.cloud.google.com
|
||||
- Try switching to the Flash model (faster, higher quota)
|
||||
- Consider using Google via OpenRouter instead for a unified quota
|
||||
|
||||
### DeepSeek
|
||||
|
||||
**Problem**: Response cuts off or streaming errors
|
||||
|
||||
**Fix**:
|
||||
- DeepSeek reasoning models (deepseek-reasoner) can produce very long outputs; try increasing Max Tokens
|
||||
- If you see streaming errors, check the DeepSeek status page
|
||||
- Try switching between deepseek-chat and deepseek-reasoner
|
||||
|
||||
---
|
||||
|
||||
## Performance Issues
|
||||
|
||||
### Slow Indexing
|
||||
|
||||
**Cause**: Large vault with many notes, or low rate limit setting.
|
||||
|
||||
**Fix**:
|
||||
- Check **Embedding Requests per Minute** — higher values speed up indexing but may cause rate limits
|
||||
- Use exclusions to skip folders you don't need indexed (e.g., large archive folders)
|
||||
- Use the incremental **Index (refresh) vault** command instead of Force Reindex when possible
|
||||
- Consider Miyo (self-host) for local indexing without API rate limits
|
||||
|
||||
### High Memory Usage
|
||||
|
||||
**Cause**: Large lexical search index or many indexed files.
|
||||
|
||||
**Fix**:
|
||||
- Reduce **Lexical Search RAM Limit** in QA settings (default 100 MB, range 20–1000 MB)
|
||||
- Add more folders to exclusions to reduce the index size
|
||||
- On mobile, disable indexing altogether
|
||||
|
||||
### UI Lag
|
||||
|
||||
**Cause**: Rendering many chat messages or a very long conversation.
|
||||
|
||||
**Fix**:
|
||||
- Start a new chat — long conversations can slow down rendering
|
||||
- Auto-compact will trigger automatically at 128,000 tokens to keep conversations manageable
|
||||
- Lower your auto-compact threshold if you're hitting performance issues early
|
||||
|
||||
---
|
||||
|
||||
## Settings Issues
|
||||
|
||||
### Reset Settings to Default
|
||||
|
||||
If your settings get into a bad state, you can reset:
|
||||
|
||||
1. Go to **Settings → Copilot** → find the reset option
|
||||
2. Or delete the `data.json` file from the plugin folder: `.obsidian/plugins/copilot/data.json`
|
||||
|
||||
⚠️ Resetting clears all your settings. API keys kept in `data.json` (standard storage) are removed, but keys stored in the Obsidian Keychain are **not** — to erase those, use **Settings → Copilot → Advanced → API Key Storage → Delete All Keys**. Back up your keys first.
|
||||
|
||||
### API Key Storage
|
||||
|
||||
Copilot has two ways to store API keys:
|
||||
|
||||
- **Standard storage**: API keys are saved in `data.json` in plain text. Existing vaults stay in this mode until you choose to migrate.
|
||||
- **Obsidian Keychain**: New installs use this by default. You can also switch an existing vault by going to **Settings → Copilot → Advanced → API Key Storage** and clicking **Migrate to Obsidian Keychain**. After migration, `data.json` no longer contains your API keys.
|
||||
|
||||
The Obsidian Keychain is per device. If you sync your vault to another device, you may need to re-enter API keys there.
|
||||
|
||||
### Debug Mode and Logs
|
||||
|
||||
For reporting bugs:
|
||||
|
||||
1. **Enable Debug Mode**: **Settings → Copilot → Advanced → Debug Mode**
|
||||
2. **Create a log file**: **Settings → Copilot → Advanced → Create Log File**
|
||||
3. The log file opens in your vault — attach it to your bug report
|
||||
|
||||
---
|
||||
|
||||
## Frequently Asked Questions
|
||||
|
||||
### Is my data private? Does Copilot send my notes to the cloud?
|
||||
|
||||
Copilot itself doesn't store your notes on any server. However, when you send a message, the content (including any context from your notes) is sent to the AI provider you've configured (OpenAI, Anthropic, etc.) via their API. Each provider has its own privacy policy. Your notes are not sent anywhere until you actively use the chat.
|
||||
|
||||
The memory system stores data in your vault locally. Chat history is saved as markdown files in your vault. Nothing is stored on Copilot's servers unless you use Copilot Plus cloud features.
|
||||
|
||||
**For maximum privacy**: Google Gemini's paid API (the basis for copilot-plus-flash) does not use API request data to train its models. For complete local privacy, consider using Ollama or LM Studio with a local model — nothing leaves your machine. Self-host mode is available now for lifetime license holders — see [Copilot Plus and Self-Host](copilot-plus-and-self-host.md) for details.
|
||||
|
||||
### Can I reference a specific note in chat?
|
||||
|
||||
Yes — use `[[Note Title]]` syntax directly in your message. Copilot adds that note's content as context in the background. You can also use @-mentions. See [Context and Mentions](context-and-mentions.md) for the full list of ways to add context.
|
||||
|
||||
### How do I make Copilot always reply in English?
|
||||
|
||||
Go to **Settings → Copilot → Advanced → Default System Prompt**, create a custom prompt, and add "Always respond in English." as an instruction. See [System Prompts](system-prompts.md).
|
||||
|
||||
### Can Copilot understand images in my notes?
|
||||
|
||||
Yes, but only with models that have **Vision** capability (shown by a vision icon in the model list). Make sure:
|
||||
1. You're using a vision-capable model
|
||||
2. **Settings → Copilot → Basic → Pass markdown images to AI** is enabled
|
||||
|
||||
### Why can't Copilot read my PDF?
|
||||
|
||||
- Large PDFs (over 10 MB) should be converted to markdown first
|
||||
- In Copilot Plus mode, use **+ Add context** to attach a PDF — it will be converted automatically
|
||||
- For large PDF collections, **Projects mode** is better suited (supports PDF as context natively)
|
||||
|
||||
### Can I use Copilot offline?
|
||||
|
||||
With local models (Ollama or LM Studio), yes — once a model is downloaded, it runs fully offline. Cloud providers (OpenAI, Anthropic, etc.) require an internet connection.
|
||||
|
||||
Lexical vault search works offline. Semantic search requires an embedding model, which may also need an internet connection unless you're using a local embedding provider or Miyo.
|
||||
|
||||
### What's the difference between Chat mode and Vault QA mode?
|
||||
|
||||
- **Chat** — General conversation. The AI only has access to your current note and anything you explicitly mention.
|
||||
- **Vault QA** — Specifically designed for asking questions about your vault. Copilot automatically searches your notes for relevant content and includes it as context.
|
||||
|
||||
For most question-and-answer tasks over your vault, use **Vault QA** or **Copilot Plus** mode.
|
||||
|
||||
### Can I use multiple providers at the same time?
|
||||
|
||||
Yes. You can have API keys configured for multiple providers simultaneously and switch between models from different providers at any time. You can even set a different model for quick commands vs. regular chat.
|
||||
|
||||
### Where are my saved chats stored?
|
||||
|
||||
Chat conversations are saved as markdown files in your vault, in the folder `copilot/copilot-conversations/` by default. You can change this folder in **Settings → Copilot → Basic → Default save folder**.
|
||||
|
||||
### How do I clear the Copilot cache?
|
||||
|
||||
Use **Command palette → Clear Copilot cache**. This clears cached responses and processed files. It does not affect your chat history or the vault index.
|
||||
|
||||
### What is the `copilot/` folder in my vault?
|
||||
|
||||
The `copilot/` folder is created by the plugin and stores:
|
||||
- `copilot-conversations/` — Saved chat histories
|
||||
- `copilot-custom-prompts/` — Your custom commands
|
||||
- `system-prompts/` — Your custom system prompts
|
||||
- `memory/` — Saved AI memories (if enabled)
|
||||
|
||||
This folder is automatically excluded from vault search to avoid cluttering results.
|
||||
|
||||
### How do I switch modes?
|
||||
|
||||
Click the mode selector at the top of the chat panel. Available modes:
|
||||
- Chat
|
||||
- Vault QA (Basic)
|
||||
- Copilot Plus (requires license)
|
||||
- Projects (alpha)
|
||||
|
||||
### The AI keeps forgetting what we talked about earlier
|
||||
|
||||
This usually means the conversation has grown too long and older turns are being trimmed from context. Options:
|
||||
- Lower **Conversation Turns in Context** in Model settings
|
||||
- Let auto-compact handle it (it summarizes old turns automatically)
|
||||
- Start a new chat and reference the previous chat file
|
||||
|
||||
---
|
||||
|
||||
## Getting More Help
|
||||
|
||||
- **GitHub Issues**: Report bugs at https://github.com/logancyang/obsidian-copilot/issues
|
||||
- **Discord**: Join the Copilot Discord community for help from other users
|
||||
- **Log file**: Create a log file (**Settings → Copilot → Advanced → Create Log File**) and include it in bug reports
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Getting Started](getting-started.md) — First-time setup
|
||||
- [LLM Providers](llm-providers.md) — Provider-specific setup details
|
||||
- [Vault Search and Indexing](vault-search-and-indexing.md) — Index management
|
||||
181
docs/vault-search-and-indexing.md
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
# Vault Search and Indexing
|
||||
|
||||
Copilot can search your vault to find relevant notes and answer questions grounded in your own content. This guide explains the two types of search, how to manage the index, and how to configure what gets indexed.
|
||||
|
||||
---
|
||||
|
||||
## Two Types of Search
|
||||
|
||||
### Lexical Search (Keyword-Based)
|
||||
|
||||
Lexical search finds notes that contain the exact words you used. It's fast, requires no setup, and works out of the box.
|
||||
|
||||
- **Used in**: Vault QA (Basic) mode
|
||||
- **How it works**: Looks for your exact keywords in note titles and content
|
||||
- **Strengths**: Fast, precise, no embedding API calls needed
|
||||
- **Limitations**: Won't find notes that use different words to express the same idea
|
||||
|
||||
**RAM Limit**: The lexical search index is held in memory. You can configure the memory limit in **Settings → Copilot → QA → Lexical Search RAM Limit** (default: 100 MB, range: 20–1,000 MB).
|
||||
|
||||
**Lexical Boosts**: Copilot can boost search results from notes in the same folder as the current note, or from notes that link to each other. Enable in **Settings → Copilot → QA → Enable Lexical Boosts** (on by default).
|
||||
|
||||
### Semantic Search (Meaning-Based)
|
||||
|
||||
Semantic search finds notes that are conceptually related, even if they don't share exact words.
|
||||
|
||||
- **Used in**: Vault QA and Copilot Plus modes — but **disabled by default**. You must explicitly enable it.
|
||||
- **How it works**: Converts your notes into numerical vectors (using an embedding model), then finds notes whose vectors are closest to your query
|
||||
- **Strengths**: Finds notes by concept and meaning, great for "fuzzy" recall
|
||||
- **Cost**: Requires embedding API calls (costs money for paid embedding models)
|
||||
- **Enable**: **Settings → Copilot → QA → Enable Semantic Search** — turn this on to activate semantic search
|
||||
|
||||
---
|
||||
|
||||
## Index Management
|
||||
|
||||
The semantic search index stores the vector embeddings of your notes. Manage it from **Settings → Copilot → QA**.
|
||||
|
||||
### Auto-Index Strategy
|
||||
|
||||
Controls when Copilot automatically updates the index:
|
||||
|
||||
| Strategy | When the index updates |
|
||||
|---|---|
|
||||
| **NEVER** | Manual only — you must trigger indexing yourself |
|
||||
| **ON STARTUP** | Updates when Obsidian starts or the plugin reloads |
|
||||
| **ON MODE SWITCH** | Updates when you switch to Vault QA or Copilot Plus mode (Recommended) |
|
||||
|
||||
The default is **ON MODE SWITCH**.
|
||||
|
||||
> **Warning**: For large vaults using paid embedding models, frequent indexing can incur significant costs. Consider using NEVER and indexing manually if cost is a concern.
|
||||
|
||||
### Refresh Index (Incremental)
|
||||
|
||||
**Command palette → Index (refresh) vault**
|
||||
|
||||
Updates only notes that have been added, modified, or deleted since the last index. Faster and cheaper than a full reindex.
|
||||
|
||||
### Force Reindex
|
||||
|
||||
**Command palette → Force reindex vault**
|
||||
|
||||
Rebuilds the entire index from scratch. Use this if:
|
||||
- You changed your embedding model
|
||||
- The index seems corrupted or missing results
|
||||
- You've made many changes and want a clean state
|
||||
|
||||
### Garbage Collection
|
||||
|
||||
**Command palette → Garbage collect Copilot index (remove files that no longer exist in vault)**
|
||||
|
||||
Removes entries from the index for notes that have been deleted from your vault. Keeps the index clean without a full reindex.
|
||||
|
||||
### Clear Index
|
||||
|
||||
**Command palette → Clear local Copilot index**
|
||||
|
||||
Deletes the entire index. You'll need to reindex before semantic search works again.
|
||||
|
||||
### Debug Commands
|
||||
|
||||
For troubleshooting:
|
||||
|
||||
- **List indexed files** — Shows all notes currently in the index
|
||||
- **Inspect index by note paths** — Check which chunks of specific notes are indexed
|
||||
- **Count total vault tokens** — Estimates total tokens across your vault
|
||||
- **Search semantic index** — Run a direct search query against the index
|
||||
|
||||
---
|
||||
|
||||
## Filtering: What Gets Indexed
|
||||
|
||||
Control which notes are included in semantic search.
|
||||
|
||||
### Cost Estimation Before Indexing
|
||||
|
||||
Before indexing a large vault with a paid embedding model, estimate the cost first:
|
||||
|
||||
**Command palette → Count total tokens in your vault**
|
||||
|
||||
This shows the total token count across your vault, which you can use to estimate embedding API costs. Embedding costs are generally low, but worth checking for very large vaults.
|
||||
|
||||
### Exclusions
|
||||
|
||||
**Settings → Copilot → QA → Exclusions**
|
||||
|
||||
Comma-separated list of patterns. Notes matching these patterns are excluded. Supports:
|
||||
- Folder names: `private` — excludes the folder named "private"
|
||||
- Folder paths: `Work/Confidential` — excludes that specific subfolder
|
||||
- File extensions: `.pdf` — excludes all PDF files
|
||||
- Tags: `#private` — excludes all notes tagged `#private`
|
||||
- Note titles: `My Secret Note` — excludes that specific note
|
||||
|
||||
Example: `private, Work/Confidential, #private` excludes the private folder, a specific work folder, and all notes tagged #private.
|
||||
|
||||
> **Note**: Tag matching works with tags in the note's **properties (frontmatter)**, not inline tags within the note body.
|
||||
|
||||
The `copilot` folder is always excluded automatically (it contains the plugin's own files).
|
||||
|
||||
### Inclusions
|
||||
|
||||
**Settings → Copilot → QA → Inclusions**
|
||||
|
||||
Comma-separated list. If set, **only** notes matching these patterns are indexed. Useful for indexing a specific area of your vault.
|
||||
|
||||
Leave empty to include everything (except exclusions).
|
||||
|
||||
---
|
||||
|
||||
## Embedding Settings
|
||||
|
||||
These settings appear in **Settings → Copilot → QA** when Semantic Search is enabled.
|
||||
|
||||
### Requests per Minute
|
||||
|
||||
How many embedding API requests to send per minute. Default is 60. Decrease this if you hit rate limit errors from your embedding provider.
|
||||
|
||||
Range: 10–60
|
||||
|
||||
### Embedding Batch Size
|
||||
|
||||
How many text chunks to send per API request. Default is 16. Larger batches are faster but may cause issues with some providers.
|
||||
|
||||
### Partitions
|
||||
|
||||
The index is split into partitions to handle large vaults. You can control the number of partitions in **Settings → Copilot → QA → Number of Partitions**. If you have a large vault, increase this value to avoid index errors.
|
||||
|
||||
> **If you hit a "RangeError: invalid string length" error**: This means your vault is too large for a single partition. Increase the number of partitions in QA settings. A good rule of thumb is that the first partition file (found in `.obsidian/`) should be under ~400 MB.
|
||||
|
||||
---
|
||||
|
||||
## Inline Citations (Experimental)
|
||||
|
||||
When enabled, AI responses in Vault QA include footnote-style citations pointing to the source notes used in the answer.
|
||||
|
||||
**Enable**: **Settings → Copilot → QA → Enable Inline Citations**
|
||||
|
||||
This is an experimental feature. Not all models handle it well.
|
||||
|
||||
---
|
||||
|
||||
## Obsidian Sync
|
||||
|
||||
If you use Obsidian Sync, the vector index can be synced across devices. Enable **Settings → Copilot → QA → Enable Index Sync**.
|
||||
|
||||
> **Note**: The index can be large (hundreds of MB for big vaults). Keep this in mind for sync limits and mobile data usage.
|
||||
|
||||
---
|
||||
|
||||
## Mobile Considerations
|
||||
|
||||
By default, Copilot **disables indexing on mobile** to save battery and data. The setting is in **Settings → Copilot → QA → Disable index on mobile** (on by default).
|
||||
|
||||
On mobile, you can still use Vault QA with lexical search, but semantic search won't update automatically.
|
||||
|
||||
---
|
||||
|
||||
## Related
|
||||
|
||||
- [Agent Mode and Tools](agent-mode-and-tools.md) — How @vault uses the index in Plus mode
|
||||
- [Models and Parameters](models-and-parameters.md) — Choosing an embedding model
|
||||
- [Copilot Plus and Self-Host](copilot-plus-and-self-host.md) — Miyo-powered local semantic search
|
||||
|
|
@ -2,11 +2,19 @@ import esbuild from "esbuild";
|
|||
import svgPlugin from "esbuild-plugin-svg";
|
||||
import process from "process";
|
||||
import wasmPlugin from "./wasmPlugin.mjs";
|
||||
import nodeModuleShim from "./nodeModuleShim.mjs";
|
||||
|
||||
const banner = `/*
|
||||
THIS IS A GENERATED/BUNDLED FILE BY ESBUILD
|
||||
if you want to view the source, please visit the github repository of this plugin
|
||||
*/
|
||||
|
||||
// Polyfill for import.meta in CommonJS context
|
||||
if (typeof import_meta === 'undefined') {
|
||||
var import_meta = {
|
||||
url: typeof __filename !== 'undefined' ? 'file://' + __filename : 'file:///obsidian-plugin'
|
||||
};
|
||||
}
|
||||
`;
|
||||
|
||||
const prod = process.argv[2] === "production";
|
||||
|
|
@ -31,6 +39,14 @@ const context = await esbuild.context({
|
|||
"@lezer/common",
|
||||
"@lezer/highlight",
|
||||
"@lezer/lr",
|
||||
// Node.js built-in modules (available in Electron) - except node:module which we shim
|
||||
"node:fs",
|
||||
"node:path",
|
||||
"node:url",
|
||||
"node:buffer",
|
||||
"node:stream",
|
||||
"node:crypto",
|
||||
"node:async_hooks",
|
||||
],
|
||||
format: "cjs",
|
||||
target: "es2020",
|
||||
|
|
@ -38,10 +54,11 @@ const context = await esbuild.context({
|
|||
sourcemap: prod ? false : "inline",
|
||||
treeShaking: true,
|
||||
outfile: "main.js",
|
||||
plugins: [svgPlugin(), wasmPlugin],
|
||||
plugins: [nodeModuleShim, svgPlugin(), wasmPlugin],
|
||||
define: {
|
||||
global: "window",
|
||||
"process.env.NODE_ENV": prod ? '"production"' : '"development"',
|
||||
"import.meta.url": "import_meta.url",
|
||||
},
|
||||
minify: prod,
|
||||
});
|
||||
|
|
|
|||
279
eslint.config.mjs
Normal file
|
|
@ -0,0 +1,279 @@
|
|||
import obsidianmd from "eslint-plugin-obsidianmd";
|
||||
import eslintReact from "@eslint-react/eslint-plugin";
|
||||
import reactHooks from "eslint-plugin-react-hooks";
|
||||
import tailwind from "eslint-plugin-tailwindcss";
|
||||
import globals from "globals";
|
||||
|
||||
export default [
|
||||
{
|
||||
ignores: [
|
||||
"node_modules/**",
|
||||
"main.js",
|
||||
"styles.css",
|
||||
"data.json",
|
||||
"designdocs/**",
|
||||
"docs/**",
|
||||
],
|
||||
},
|
||||
|
||||
// obsidianmd recommended brings:
|
||||
// - eslint:recommended
|
||||
// - typescript-eslint recommendedTypeChecked on .ts/.tsx (recommended on .js/.jsx)
|
||||
// - obsidianmd plugin + all obsidianmd-namespaced rules
|
||||
// - import / @microsoft/sdl / depend / no-unsanitized
|
||||
// - Obsidian-injected globals (activeDocument, createDiv, etc.)
|
||||
...obsidianmd.configs.recommended,
|
||||
|
||||
// React + tailwind plugins ship flat configs with no `files` filter, so
|
||||
// they'd cascade onto package.json (which uses the JSON parser) and crash.
|
||||
// Constrain them to JSX/TSX sources where React/JSX rules actually apply.
|
||||
{
|
||||
files: ["**/*.{jsx,tsx}"],
|
||||
...eslintReact.configs.recommended,
|
||||
},
|
||||
{
|
||||
files: ["**/*.{jsx,tsx}"],
|
||||
rules: {
|
||||
// Deferred to follow-up PRs — these flag legitimate anti-patterns but
|
||||
// each fix requires per-component intent analysis, and they're surfaced
|
||||
// as warnings (not errors) so they don't block CI.
|
||||
//
|
||||
// no-direct-set-state-in-use-effect: ~50 violations. Common pattern is
|
||||
// "sync local state with prop", which has no one-size-fits-all fix —
|
||||
// some cases want render-time derivation, others want a `key` prop reset
|
||||
// or `useSyncExternalStore`. Refactoring blindly risks behavior regressions
|
||||
// in the chat UI's stateful components.
|
||||
"@eslint-react/hooks-extra/no-direct-set-state-in-use-effect": "warn",
|
||||
},
|
||||
},
|
||||
{
|
||||
files: ["**/*.{js,jsx,mjs,cjs,ts,tsx}"],
|
||||
plugins: { "react-hooks": reactHooks },
|
||||
rules: {
|
||||
"react-hooks/rules-of-hooks": "error",
|
||||
"react-hooks/exhaustive-deps": "error",
|
||||
},
|
||||
},
|
||||
...tailwind.configs["flat/recommended"].map((cfg) => ({
|
||||
files: ["**/*.{js,jsx,mjs,cjs,ts,tsx}"],
|
||||
...cfg,
|
||||
})),
|
||||
|
||||
{
|
||||
files: ["**/*.{js,jsx,mjs,cjs,ts,tsx}"],
|
||||
languageOptions: {
|
||||
globals: {
|
||||
// Obsidian plugin runtime injects `app` as a global (see CLAUDE.md).
|
||||
app: "readonly",
|
||||
},
|
||||
},
|
||||
settings: {
|
||||
"react-x": { version: "detect" },
|
||||
tailwindcss: {
|
||||
callees: ["classnames", "clsx", "ctl", "cn", "cva"],
|
||||
config: "./tailwind.config.js",
|
||||
cssFiles: ["**/*.css", "!**/node_modules", "!**/.*", "!**/dist", "!**/build"],
|
||||
// Obsidian-provided utility classes used in JSX but not defined in our CSS.
|
||||
whitelist: ["clickable-icon"],
|
||||
},
|
||||
},
|
||||
rules: {
|
||||
// Carry-over from legacy .eslintrc
|
||||
"no-prototype-builtins": "off",
|
||||
"tailwindcss/classnames-order": "error",
|
||||
"tailwindcss/enforces-negative-arbitrary-values": "error",
|
||||
"tailwindcss/enforces-shorthand": "error",
|
||||
"tailwindcss/migration-from-tailwind-2": "error",
|
||||
"tailwindcss/no-arbitrary-value": "off",
|
||||
"tailwindcss/no-custom-classname": "error",
|
||||
"tailwindcss/no-contradicting-classname": "error",
|
||||
|
||||
// obsidianmd: defer to follow-up PRs
|
||||
"obsidianmd/ui/sentence-case": "off",
|
||||
|
||||
// obsidianmd: disabled intentionally — Platform.isMacOS branching is on-purpose
|
||||
"obsidianmd/platform": "off",
|
||||
|
||||
// Bundled by obsidianmd/recommended via tseslint.configs.recommendedTypeChecked.
|
||||
// Disabled here because the codebase intentionally uses `any` / dynamic typing
|
||||
// around Obsidian's untyped APIs and LangChain message shapes — flipping these
|
||||
// on would require refactoring thousands of call sites with no functional gain.
|
||||
//
|
||||
// Violation counts (src/**/*.{ts,tsx}) are noted inline. Rules with low counts
|
||||
// are candidates to enable in small follow-up PRs.
|
||||
|
||||
// --- Heavy: any-flow through Obsidian/LangChain APIs ---
|
||||
// no-unsafe-member-access: enabled globally; tests are exempted via the
|
||||
// test-file override below.
|
||||
"@typescript-eslint/no-unsafe-assignment": "off", // enabled for tests below; follow-up PR for production
|
||||
"@typescript-eslint/no-unsafe-call": "off", // 107 violations
|
||||
|
||||
// --- Medium: promise / method ergonomics ---
|
||||
// Enabled in the TS-only block below.
|
||||
|
||||
// no-deprecated: defer — surface the warnings, but don't fail CI yet
|
||||
"@typescript-eslint/no-deprecated": "off",
|
||||
|
||||
// SDL / import / no-unsanitized / depend: defer — review separately
|
||||
"no-restricted-globals": "off",
|
||||
},
|
||||
},
|
||||
|
||||
// Guardrail: every standalone React root in the plugin must go through
|
||||
// `createPluginRoot` so descendants can rely on `useApp()` unconditionally
|
||||
// (the bug class fixed in PR #2466). Forbid importing `createRoot` from
|
||||
// `react-dom/client` anywhere except the helper itself.
|
||||
{
|
||||
files: ["src/**/*.{ts,tsx}"],
|
||||
ignores: ["src/utils/react/createPluginRoot.tsx"],
|
||||
rules: {
|
||||
"no-restricted-syntax": [
|
||||
"error",
|
||||
{
|
||||
selector:
|
||||
"ImportDeclaration[source.value='react-dom/client'] ImportSpecifier[imported.name='createRoot']",
|
||||
message:
|
||||
"Use createPluginRoot from '@/utils/react/createPluginRoot' instead. It wraps the root in <AppContext.Provider> so descendants can rely on useApp() unconditionally (see PR #2466).",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
// Test files need Jest globals
|
||||
{
|
||||
files: ["**/*.test.{js,jsx,ts,tsx}", "jest.setup.js", "__mocks__/**"],
|
||||
languageOptions: {
|
||||
globals: {
|
||||
...globals.jest,
|
||||
...globals.node,
|
||||
},
|
||||
},
|
||||
rules: {
|
||||
"import/no-nodejs-modules": "off",
|
||||
// Tests use intentional `any` mocks; disable type-safety rules that flood
|
||||
// the test suite without adding signal.
|
||||
"@typescript-eslint/no-unsafe-member-access": "off",
|
||||
},
|
||||
},
|
||||
|
||||
// Tests have been cleaned of unsafe `any` assignments. Production code
|
||||
// (~499 violations) is a follow-up; keep tests enforced.
|
||||
{
|
||||
files: ["**/*.test.{ts,tsx}"],
|
||||
rules: {
|
||||
"@typescript-eslint/no-unsafe-assignment": "error",
|
||||
},
|
||||
},
|
||||
|
||||
// Integration tests bootstrap jsdom fetch via `node-fetch` polyfill —
|
||||
// allow the otherwise-banned import here only.
|
||||
{
|
||||
files: ["src/integration_tests/**"],
|
||||
rules: {
|
||||
"no-restricted-imports": "off",
|
||||
},
|
||||
},
|
||||
|
||||
// Node-context files (build configs, scripts)
|
||||
{
|
||||
files: [
|
||||
"*.{js,mjs,cjs}",
|
||||
"scripts/**",
|
||||
"esbuild.config.mjs",
|
||||
"version-bump.mjs",
|
||||
"wasmPlugin.mjs",
|
||||
"nodeModuleShim.mjs",
|
||||
"jest.config.js",
|
||||
"tailwind.config.js",
|
||||
],
|
||||
languageOptions: {
|
||||
globals: {
|
||||
...globals.node,
|
||||
},
|
||||
},
|
||||
rules: {
|
||||
"import/no-nodejs-modules": "off",
|
||||
},
|
||||
},
|
||||
|
||||
// TypeScript-specific overrides (the @typescript-eslint plugin is registered
|
||||
// by obsidianmd's recommended config only for .ts/.tsx files).
|
||||
{
|
||||
files: ["**/*.ts", "**/*.tsx"],
|
||||
languageOptions: {
|
||||
parserOptions: {
|
||||
project: "./tsconfig.json",
|
||||
tsconfigRootDir: import.meta.dirname,
|
||||
},
|
||||
},
|
||||
rules: {
|
||||
"@typescript-eslint/no-empty-function": "off",
|
||||
"@typescript-eslint/ban-ts-comment": "off",
|
||||
"@typescript-eslint/no-unused-vars": ["error", { args: "none" }],
|
||||
// checksVoidReturn relaxed for:
|
||||
// - attributes: async event handlers in JSX (onClick={async () => ...}) are
|
||||
// the standard React pattern; React already handles them correctly.
|
||||
// - inheritedMethods: Obsidian's Plugin.onload/onunload are commonly async.
|
||||
"@typescript-eslint/no-misused-promises": [
|
||||
"error",
|
||||
{ checksVoidReturn: { attributes: false, inheritedMethods: false } },
|
||||
],
|
||||
"@typescript-eslint/no-floating-promises": "error",
|
||||
"@typescript-eslint/no-unsafe-return": "error",
|
||||
"@typescript-eslint/unbound-method": "error",
|
||||
// TypeScript handles undefined-identifier detection (and does so cross-realm
|
||||
// correctly); per typescript-eslint's own guidance, disable no-undef on TS.
|
||||
"no-undef": "off",
|
||||
},
|
||||
},
|
||||
|
||||
// Non-TS files aren't in tsconfig.json — disable type-aware rules that
|
||||
// obsidianmd's recommended config enables globally. Most typed obsidianmd
|
||||
// rules are already gated to **/*.ts(x); only no-plugin-as-component leaks
|
||||
// out via recommendedPluginRulesConfig, and @typescript-eslint/no-deprecated
|
||||
// is enabled globally.
|
||||
{
|
||||
files: ["**/*.js", "**/*.mjs", "**/*.cjs", "**/*.jsx", "**/package.json"],
|
||||
rules: {
|
||||
"@typescript-eslint/no-deprecated": "off",
|
||||
"obsidianmd/no-plugin-as-component": "off",
|
||||
},
|
||||
},
|
||||
|
||||
// package.json: keep depend/ban-dependencies enabled (from obsidianmd
|
||||
// recommended) but allow the deps we deliberately keep.
|
||||
{
|
||||
files: ["**/package.json"],
|
||||
rules: {
|
||||
"depend/ban-dependencies": [
|
||||
"error",
|
||||
{
|
||||
presets: ["native", "microutilities", "preferred"],
|
||||
allowed: [],
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
// logger.ts is the central logging utility and must call console.* directly.
|
||||
// scripts/** are CLI tools that print to stdout.
|
||||
{
|
||||
files: ["src/logger.ts", "scripts/**"],
|
||||
rules: {
|
||||
"obsidianmd/rule-custom-message": "off",
|
||||
},
|
||||
},
|
||||
|
||||
// Jest assertions like `expect(mock.method).toHaveBeenCalled()` reference
|
||||
// methods unbound by design. The rule has no clean workaround for jest
|
||||
// patterns (binding changes the reference identity and breaks the assertion),
|
||||
// so disable it in tests. Scoped to .ts/.tsx because the @typescript-eslint
|
||||
// plugin is only registered for those files. Placed last so it overrides the
|
||||
// TS-only block above.
|
||||
{
|
||||
files: ["**/*.test.{ts,tsx}"],
|
||||
rules: {
|
||||
"@typescript-eslint/unbound-method": "off",
|
||||
},
|
||||
},
|
||||
];
|
||||
|
Before Width: | Height: | Size: 680 KiB |
BIN
images/Add-Context.png
Normal file
|
After Width: | Height: | Size: 801 KiB |
BIN
images/Add-Selection-to-Context.png
Normal file
|
After Width: | Height: | Size: 1.1 MiB |
BIN
images/Agent-Mode.png
Normal file
|
After Width: | Height: | Size: 917 KiB |
|
Before Width: | Height: | Size: 723 KiB After Width: | Height: | Size: 1.1 MiB |
|
Before Width: | Height: | Size: 210 KiB |
BIN
images/Create-Command.png
Normal file
|
After Width: | Height: | Size: 392 KiB |
BIN
images/Note-Image.png
Normal file
|
After Width: | Height: | Size: 1.7 MiB |
|
Before Width: | Height: | Size: 430 KiB After Width: | Height: | Size: 516 KiB |
|
Before Width: | Height: | Size: 1.1 MiB After Width: | Height: | Size: 1.1 MiB |
|
Before Width: | Height: | Size: 532 KiB After Width: | Height: | Size: 855 KiB |
BIN
images/Quick-Command.png
Normal file
|
After Width: | Height: | Size: 487 KiB |
|
Before Width: | Height: | Size: 626 KiB After Width: | Height: | Size: 1.1 MiB |
|
Before Width: | Height: | Size: 595 KiB After Width: | Height: | Size: 940 KiB |
|
Before Width: | Height: | Size: 243 KiB After Width: | Height: | Size: 963 KiB |
BIN
images/chat mode image.png
Normal file
|
After Width: | Height: | Size: 541 KiB |
|
Before Width: | Height: | Size: 367 KiB |
BIN
images/product-ui-screenshot.png
Normal file
|
After Width: | Height: | Size: 1.7 MiB |
|
|
@ -6,9 +6,12 @@ module.exports = {
|
|||
"^.+\\.(js|jsx|ts|tsx)$": "ts-jest",
|
||||
},
|
||||
moduleNameMapper: {
|
||||
"\\.(css|less|scss|sass)$": "identity-obj-proxy",
|
||||
"^@/(.*)$": "<rootDir>/src/$1",
|
||||
"^obsidian$": "<rootDir>/__mocks__/obsidian.js",
|
||||
// The yaml package's "exports" field defaults to a browser ESM entry under
|
||||
// jsdom; Jest can't parse ESM without extra config, so point at the CJS
|
||||
// build it ships under dist/.
|
||||
"^yaml$": "<rootDir>/node_modules/yaml/dist/index.js",
|
||||
},
|
||||
testRegex: ".*\\.test\\.(jsx?|tsx?)$",
|
||||
moduleFileExtensions: ["ts", "tsx", "js", "jsx", "json", "node"],
|
||||
|
|
|
|||
|
|
@ -1,5 +1,24 @@
|
|||
import "web-streams-polyfill/dist/polyfill.min.js";
|
||||
import { TextEncoder, TextDecoder } from "util";
|
||||
|
||||
global.TextEncoder = TextEncoder;
|
||||
global.TextDecoder = TextDecoder;
|
||||
window.TextEncoder = TextEncoder;
|
||||
window.TextDecoder = TextDecoder;
|
||||
|
||||
// Polyfill Obsidian's Node.doc / Node.win augmentation so plugin code that
|
||||
// reads `element.doc` / `element.win` works under jsdom.
|
||||
if (typeof Node !== "undefined" && !Object.prototype.hasOwnProperty.call(Node.prototype, "doc")) {
|
||||
Object.defineProperty(Node.prototype, "doc", {
|
||||
get() {
|
||||
return this.ownerDocument ?? window.document;
|
||||
},
|
||||
configurable: true,
|
||||
});
|
||||
}
|
||||
if (typeof Node !== "undefined" && !Object.prototype.hasOwnProperty.call(Node.prototype, "win")) {
|
||||
Object.defineProperty(Node.prototype, "win", {
|
||||
get() {
|
||||
return this.ownerDocument?.defaultView ?? window;
|
||||
},
|
||||
configurable: true,
|
||||
});
|
||||
}
|
||||
|
|
|
|||
7
knip.json
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
{
|
||||
"$schema": "https://unpkg.com/knip@5/schema.json",
|
||||
"entry": ["src/main.ts", "scripts/printPromptDebug.js", "scripts/printPromptDebugEntry.ts"],
|
||||
"project": ["src/**/*.{ts,tsx,js,jsx}", "scripts/**/*.{ts,js,mjs}"],
|
||||
"ignore": ["src/styles/tailwind.css", "src/integration_tests/**"],
|
||||
"ignoreDependencies": ["buffer"]
|
||||
}
|
||||
|
|
@ -1,8 +1,9 @@
|
|||
{
|
||||
"id": "copilot",
|
||||
"name": "Copilot",
|
||||
"version": "2.9.2",
|
||||
"minAppVersion": "0.15.0",
|
||||
"version": "3.3.3",
|
||||
"minAppVersion": "1.11.4",
|
||||
"isDesktopOnly": false,
|
||||
"description": "Your AI Copilot: Chat with Your Second Brain, Learn Faster, Work Smarter.",
|
||||
"author": "Logan Yang",
|
||||
"authorUrl": "https://twitter.com/logancyang",
|
||||
|
|
|
|||
37
nodeModuleShim.mjs
Normal file
|
|
@ -0,0 +1,37 @@
|
|||
// Plugin to provide a shim for node:module in browser/Electron renderer context
|
||||
const nodeModuleShim = {
|
||||
name: "node-module-shim",
|
||||
setup(build) {
|
||||
// Intercept node:module imports and provide a shim
|
||||
build.onResolve({ filter: /^node:module$/ }, (args) => {
|
||||
return {
|
||||
path: args.path,
|
||||
namespace: "node-module-shim",
|
||||
};
|
||||
});
|
||||
|
||||
build.onLoad({ filter: /.*/, namespace: "node-module-shim" }, () => {
|
||||
return {
|
||||
contents: `
|
||||
// Shim for node:module in Electron/Obsidian environment (CommonJS format)
|
||||
module.exports = {
|
||||
createRequire: function(filename) {
|
||||
// In Electron renderer, we can use the global require
|
||||
// Note: filename parameter is ignored (may be undefined from @langchain/community v1.0.0)
|
||||
if (typeof require !== 'undefined') {
|
||||
return require;
|
||||
}
|
||||
// Fallback: return a function that throws a helpful error
|
||||
return function shimmedRequire(id) {
|
||||
throw new Error('Dynamic require of "' + id + '" is not supported in this environment');
|
||||
};
|
||||
}
|
||||
};
|
||||
`,
|
||||
loader: "js",
|
||||
};
|
||||
});
|
||||
},
|
||||
};
|
||||
|
||||
export default nodeModuleShim;
|
||||
14533
package-lock.json
generated
134
package.json
|
|
@ -1,143 +1,125 @@
|
|||
{
|
||||
"name": "obsidian-copilot",
|
||||
"version": "2.9.2",
|
||||
"version": "3.3.3",
|
||||
"description": "Your AI Copilot: Chat with Your Second Brain, Learn Faster, Work Smarter.",
|
||||
"main": "main.js",
|
||||
"scripts": {
|
||||
"dev": "npm-run-all --parallel dev:*",
|
||||
"dev": "run-p dev:*",
|
||||
"dev:tailwind": "npx tailwindcss -i src/styles/tailwind.css -o styles.css --watch",
|
||||
"dev:esbuild": "node esbuild.config.mjs",
|
||||
"build": "npm run build:tailwind && npm run build:esbuild || exit 1",
|
||||
"build:tailwind": "npx tailwindcss -i src/styles/tailwind.css -o styles.css --minify",
|
||||
"build:esbuild": "tsc -noEmit -skipLibCheck && node esbuild.config.mjs production",
|
||||
"lint": "eslint . --ext .js,.jsx,.ts,.tsx",
|
||||
"lint:fix": "eslint . --ext .js,.jsx,.ts,.tsx --fix",
|
||||
"lint": "eslint .",
|
||||
"lint:dead": "npx --yes knip@5",
|
||||
"lint:fix": "eslint . --fix",
|
||||
"format": "prettier --write 'src/**/*.{js,ts,tsx,md}'",
|
||||
"format:check": "prettier --check 'src/**/*.{js,ts,tsx,md}'",
|
||||
"version": "node version-bump.mjs && git add manifest.json versions.json",
|
||||
"version": "node version-bump.mjs",
|
||||
"test": "jest --testPathIgnorePatterns=src/integration_tests/",
|
||||
"test:integration": "jest src/integration_tests/",
|
||||
"prepare": "husky"
|
||||
"prepare": "husky",
|
||||
"prompt:debug": "node scripts/printPromptDebug.js",
|
||||
"test:vault": "bash scripts/test-vault.sh"
|
||||
},
|
||||
"keywords": [],
|
||||
"author": "Logan Yang",
|
||||
"husky": {
|
||||
"hooks": {
|
||||
"pre-commit": "lint-staged"
|
||||
}
|
||||
},
|
||||
"lint-staged": {
|
||||
"nano-staged": {
|
||||
"*.{js,jsx,ts,tsx,json,css,md}": [
|
||||
"prettier --write",
|
||||
"git add"
|
||||
"prettier --write"
|
||||
],
|
||||
"*.{js,jsx,ts,tsx}": [
|
||||
"eslint --fix"
|
||||
]
|
||||
},
|
||||
"license": "AGPL-3.0",
|
||||
"devDependencies": {
|
||||
"@langchain/ollama": "^0.2.0",
|
||||
"@codemirror/state": "^6.5.2",
|
||||
"@codemirror/view": "^6.36.4",
|
||||
"@eslint-react/eslint-plugin": "^1.38.4",
|
||||
"@google/generative-ai": "^0.24.0",
|
||||
"@jest/globals": "^29.7.0",
|
||||
"@langchain/ollama": "^1.2.2",
|
||||
"@tailwindcss/container-queries": "^0.1.1",
|
||||
"@testing-library/jest-dom": "^5.16.5",
|
||||
"@testing-library/react": "^14.0.0",
|
||||
"@types/crypto-js": "^4.1.1",
|
||||
"@types/diff": "^7.0.1",
|
||||
"@types/events": "^3.0.0",
|
||||
"@types/jest": "^29.5.11",
|
||||
"@types/koa": "^2.13.7",
|
||||
"@types/koa__cors": "^4.0.0",
|
||||
"@types/lodash.debounce": "^4.0.9",
|
||||
"@types/luxon": "^3.4.2",
|
||||
"@types/node": "^16.11.6",
|
||||
"@types/node-fetch": "^2.6.12",
|
||||
"@types/react": "^18.0.33",
|
||||
"@types/react-dom": "^18.0.11",
|
||||
"@types/react-syntax-highlighter": "^15.5.6",
|
||||
"@typescript-eslint/eslint-plugin": "^8.19.1",
|
||||
"@typescript-eslint/parser": "^8.19.1",
|
||||
"builtin-modules": "3.3.0",
|
||||
"@types/turndown": "^5.0.6",
|
||||
"@types/uuid": "^10.0.0",
|
||||
"electron": "^27.3.2",
|
||||
"esbuild": "^0.25.0",
|
||||
"eslint": "^8.57.0",
|
||||
"eslint-plugin-json": "^4.0.1",
|
||||
"eslint-plugin-react": "^7.37.3",
|
||||
"esbuild-plugin-svg": "^0.1.0",
|
||||
"eslint": "^9.18.0",
|
||||
"eslint-plugin-obsidianmd": "^0.3.0",
|
||||
"eslint-plugin-react-hooks": "^5.1.0",
|
||||
"eslint-plugin-tailwindcss": "^3.18.0",
|
||||
"globals": "^15.14.0",
|
||||
"husky": "^9.1.5",
|
||||
"jest": "^29.7.0",
|
||||
"jest-environment-jsdom": "^29.5.0",
|
||||
"lint-staged": "^15.2.9",
|
||||
"node-fetch": "^2.7.0",
|
||||
"npm-run-all": "^4.1.5",
|
||||
"nano-staged": "^0.8.0",
|
||||
"npm-run-all2": "^7.0.2",
|
||||
"obsidian": "^1.2.5",
|
||||
"prettier": "^3.3.3",
|
||||
"tailwindcss": "^3.4.15",
|
||||
"tailwindcss-animate": "^1.0.7",
|
||||
"ts-jest": "^29.1.0",
|
||||
"tslib": "2.4.0",
|
||||
"typescript": "^5.7.2",
|
||||
"web-streams-polyfill": "^3.3.2"
|
||||
"web-streams-polyfill": "^3.3.2",
|
||||
"yaml": "^2.6.1"
|
||||
},
|
||||
"dependencies": {
|
||||
"@dnd-kit/core": "^6.3.1",
|
||||
"@dnd-kit/sortable": "^10.0.0",
|
||||
"@dnd-kit/utilities": "^3.2.2",
|
||||
"@google/generative-ai": "^0.21.0",
|
||||
"@huggingface/inference": "^2.6.4",
|
||||
"@koa/cors": "^5.0.0",
|
||||
"@langchain/anthropic": "^0.3.20",
|
||||
"@langchain/cohere": "^0.3.0",
|
||||
"@langchain/community": "^0.3.22",
|
||||
"@langchain/core": "^0.3.45",
|
||||
"@langchain/deepseek": "^0.0.1",
|
||||
"@langchain/google-genai": "^0.1.6",
|
||||
"@langchain/groq": "^0.1.2",
|
||||
"@langchain/mistralai": "^0.2.0",
|
||||
"@langchain/openai": "^0.5.6",
|
||||
"@langchain/xai": "^0.0.2",
|
||||
"@langchain/anthropic": "^1.3.29",
|
||||
"@langchain/classic": "^1.0.9",
|
||||
"@langchain/core": "^1.1.29",
|
||||
"@langchain/deepseek": "^1.0.0",
|
||||
"@langchain/google-genai": "^2.1.23",
|
||||
"@langchain/groq": "^1.0.0",
|
||||
"@langchain/openai": "^1.0.0",
|
||||
"@langchain/textsplitters": "^1.0.0",
|
||||
"@langchain/xai": "^1.0.0",
|
||||
"@lexical/react": "^0.34.0",
|
||||
"@orama/orama": "^3.0.0-rc-2",
|
||||
"@radix-ui/react-checkbox": "^1.1.3",
|
||||
"@radix-ui/react-checkbox": "^1.3.3",
|
||||
"@radix-ui/react-collapsible": "^1.1.2",
|
||||
"@radix-ui/react-dialog": "^1.1.3",
|
||||
"@radix-ui/react-dialog": "^1.1.15",
|
||||
"@radix-ui/react-dropdown-menu": "^2.1.4",
|
||||
"@radix-ui/react-label": "^2.1.0",
|
||||
"@radix-ui/react-popover": "^1.1.4",
|
||||
"@radix-ui/react-scroll-area": "^1.2.9",
|
||||
"@radix-ui/react-select": "^2.1.2",
|
||||
"@radix-ui/react-label": "^2.1.7",
|
||||
"@radix-ui/react-popover": "^1.1.15",
|
||||
"@radix-ui/react-progress": "^1.1.7",
|
||||
"@radix-ui/react-scroll-area": "^1.2.10",
|
||||
"@radix-ui/react-select": "^2.2.6",
|
||||
"@radix-ui/react-separator": "^1.1.7",
|
||||
"@radix-ui/react-slider": "^1.2.1",
|
||||
"@radix-ui/react-slot": "^1.2.3",
|
||||
"@radix-ui/react-switch": "^1.1.1",
|
||||
"@radix-ui/react-tabs": "^1.1.3",
|
||||
"@radix-ui/react-tooltip": "^1.2.7",
|
||||
"@tabler/icons-react": "^2.14.0",
|
||||
"@radix-ui/react-slider": "^1.3.5",
|
||||
"@radix-ui/react-slot": "^1.2.4",
|
||||
"@radix-ui/react-tooltip": "^1.2.8",
|
||||
"async-mutex": "^0.5.0",
|
||||
"axios": "^1.3.4",
|
||||
"buffer": "^6.0.3",
|
||||
"chrono-node": "^2.7.7",
|
||||
"class-variance-authority": "^0.7.1",
|
||||
"clsx": "^2.1.1",
|
||||
"codemirror-companion-extension": "^0.0.11",
|
||||
"cohere-ai": "^7.13.0",
|
||||
"crypto-js": "^4.1.1",
|
||||
"diff": "^7.0.0",
|
||||
"esbuild-plugin-svg": "^0.1.0",
|
||||
"eventsource-parser": "^1.0.0",
|
||||
"fuzzysort": "^3.1.0",
|
||||
"jotai": "^2.10.3",
|
||||
"koa": "^2.14.2",
|
||||
"koa-proxies": "^0.12.3",
|
||||
"langchain": "^0.3.2",
|
||||
"lodash.debounce": "^4.0.8",
|
||||
"lexical": "^0.34.0",
|
||||
"lucide-react": "^0.462.0",
|
||||
"luxon": "^3.5.0",
|
||||
"next-i18next": "^13.2.2",
|
||||
"prop-types": "^15.8.1",
|
||||
"minisearch": "^7.2.0",
|
||||
"openai": "^6.10.0",
|
||||
"react": "^18.2.0",
|
||||
"react-dom": "^18.2.0",
|
||||
"react-dropzone": "^14.3.5",
|
||||
"react-markdown": "^9.0.1",
|
||||
"react-resizable-panels": "^3.0.2",
|
||||
"react-syntax-highlighter": "^15.5.0",
|
||||
"sse": "github:mpetazzoni/sse.js",
|
||||
"tailwind-merge": "^2.5.5",
|
||||
"tailwindcss-animate": "^1.0.7",
|
||||
"trie-search": "^2.2.0"
|
||||
"turndown": "^7.2.2",
|
||||
"uuid": "^11.1.0",
|
||||
"zod": "^3.25.76"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
62
scripts/printPromptDebug.js
Executable file
|
|
@ -0,0 +1,62 @@
|
|||
#!/usr/bin/env node
|
||||
|
||||
/**
|
||||
* Bundle the TypeScript entry file into a temporary ESM module and execute it.
|
||||
*/
|
||||
async function main() {
|
||||
const [{ build }, fs, os, path, url] = await Promise.all([
|
||||
import("esbuild"),
|
||||
import("node:fs/promises"),
|
||||
import("node:os"),
|
||||
import("node:path"),
|
||||
import("node:url"),
|
||||
]);
|
||||
|
||||
const entryFile = path.resolve(__dirname, "printPromptDebugEntry.ts");
|
||||
const outfile = path.join(os.tmpdir(), `prompt-debug-${Date.now()}.mjs`);
|
||||
|
||||
if (!Array.prototype.contains) {
|
||||
Object.defineProperty(Array.prototype, "contains", {
|
||||
value(value) {
|
||||
return this.includes(value);
|
||||
},
|
||||
enumerable: false,
|
||||
});
|
||||
}
|
||||
|
||||
const obsidianStubPlugin = {
|
||||
name: "obsidian-stub",
|
||||
setup(build) {
|
||||
build.onResolve({ filter: /^obsidian$/ }, () => ({
|
||||
path: path.resolve(__dirname, "stubs/obsidian.ts"),
|
||||
}));
|
||||
},
|
||||
};
|
||||
|
||||
await build({
|
||||
entryPoints: [entryFile],
|
||||
outfile,
|
||||
bundle: true,
|
||||
platform: "node",
|
||||
format: "esm",
|
||||
sourcemap: false,
|
||||
target: "node18",
|
||||
tsconfig: path.resolve(__dirname, "../tsconfig.json"),
|
||||
plugins: [obsidianStubPlugin],
|
||||
});
|
||||
|
||||
try {
|
||||
// eslint-disable-next-line no-unsanitized/method -- outfile is a controlled path under os.tmpdir(), produced by the esbuild step above.
|
||||
const module = await import(url.pathToFileURL(outfile).href);
|
||||
await module.run(process.argv.slice(2));
|
||||
} finally {
|
||||
await fs.unlink(outfile).catch(() => {
|
||||
/* ignore cleanup errors */
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
main().catch((error) => {
|
||||
console.error("Failed to generate prompt debug report:", error);
|
||||
process.exitCode = 1;
|
||||
});
|
||||
134
scripts/printPromptDebugEntry.ts
Normal file
|
|
@ -0,0 +1,134 @@
|
|||
import type ChainManager from "@/LLMProviders/chainManager";
|
||||
import MemoryManager from "@/LLMProviders/memoryManager";
|
||||
import { ModelAdapterFactory } from "@/LLMProviders/chainRunner/utils/modelAdapter";
|
||||
import { buildAgentPromptDebugReport } from "@/LLMProviders/chainRunner/utils/promptDebugService";
|
||||
import { ToolRegistry } from "@/tools/ToolRegistry";
|
||||
import { ChatMessage } from "@/types/message";
|
||||
import { initializeBuiltinTools } from "@/tools/builtinTools";
|
||||
import { getSettings } from "@/settings/model";
|
||||
import { UserMemoryManager } from "@/memory/UserMemoryManager";
|
||||
import type { App } from "obsidian";
|
||||
import type { BaseChatModel } from "@langchain/core/language_models/chat_models";
|
||||
|
||||
interface HeadlessApp {
|
||||
vault: {
|
||||
getRoot: () => { name: string };
|
||||
getAbstractFileByPath: (path: string) => null;
|
||||
read: (file: unknown) => Promise<string>;
|
||||
getMarkdownFiles: () => unknown[];
|
||||
getAllLoadedFiles: () => unknown[];
|
||||
adapter: {
|
||||
mkdir: (path: string) => Promise<void>;
|
||||
};
|
||||
};
|
||||
metadataCache: {
|
||||
getFirstLinkpathDest: () => null;
|
||||
getFileCache: () => null;
|
||||
};
|
||||
workspace: {
|
||||
getActiveFile: () => null;
|
||||
getLeaf: () => { openFile: () => Promise<void> };
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a minimal Obsidian app stub suitable for CLI usage.
|
||||
*
|
||||
* The autonomous agent only needs vault lookups and metadata cache reads, so this
|
||||
* provides no-op implementations that satisfy those expectations.
|
||||
*/
|
||||
function createHeadlessApp(): HeadlessApp {
|
||||
return {
|
||||
vault: {
|
||||
getRoot: () => ({ name: "root" }),
|
||||
getAbstractFileByPath: () => null,
|
||||
read: async () => "",
|
||||
getMarkdownFiles: () => [],
|
||||
getAllLoadedFiles: () => [],
|
||||
adapter: {
|
||||
mkdir: async () => {
|
||||
/* no-op */
|
||||
},
|
||||
},
|
||||
},
|
||||
metadataCache: {
|
||||
getFirstLinkpathDest: () => null,
|
||||
getFileCache: () => null,
|
||||
},
|
||||
workspace: {
|
||||
getActiveFile: () => null,
|
||||
getLeaf: () => ({
|
||||
openFile: async () => {
|
||||
/* no-op */
|
||||
},
|
||||
}),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a plain user message into the ChatMessage shape used by the agent.
|
||||
*
|
||||
* @param message - Raw user text to analyse.
|
||||
* @returns Minimal chat message.
|
||||
*/
|
||||
function buildChatMessage(message: string): ChatMessage {
|
||||
return {
|
||||
message,
|
||||
originalMessage: message,
|
||||
sender: "user",
|
||||
timestamp: null,
|
||||
isVisible: true,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate the annotated prompt debug report for a given user input.
|
||||
*
|
||||
* @param args - CLI arguments (expects the user prompt as the concatenated string).
|
||||
*/
|
||||
export async function run(args: string[]): Promise<void> {
|
||||
const userInput = args.join(" ").trim();
|
||||
|
||||
if (!userInput) {
|
||||
console.error('Usage: npm run prompt:debug -- "your message here"');
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
const app = createHeadlessApp();
|
||||
// eslint-disable-next-line obsidianmd/no-global-this -- node-only debug script, no window available
|
||||
(global as unknown as { app: unknown }).app = app;
|
||||
|
||||
initializeBuiltinTools();
|
||||
|
||||
const registry = ToolRegistry.getInstance();
|
||||
const settings = getSettings();
|
||||
const enabledToolIds = new Set(settings.autonomousAgentEnabledToolIds || []);
|
||||
const availableTools = registry.getEnabledTools(enabledToolIds, false);
|
||||
|
||||
// Generate simple tool descriptions (native tool calling handles schema via bindTools)
|
||||
const toolDescriptions = availableTools
|
||||
.map((tool) => `${tool.name}: ${tool.description}`)
|
||||
.join("\n");
|
||||
|
||||
const memoryManager = MemoryManager.getInstance();
|
||||
const userMemoryManager = new UserMemoryManager(app as unknown as App);
|
||||
const chainContext = {
|
||||
memoryManager,
|
||||
userMemoryManager,
|
||||
} as unknown as ChainManager;
|
||||
|
||||
const adapter = ModelAdapterFactory.createAdapter({
|
||||
modelName: "gpt-4",
|
||||
} as unknown as BaseChatModel);
|
||||
const report = await buildAgentPromptDebugReport({
|
||||
chainManager: chainContext,
|
||||
adapter,
|
||||
availableTools,
|
||||
toolDescriptions,
|
||||
userMessage: buildChatMessage(userInput),
|
||||
});
|
||||
|
||||
console.log(report.annotatedPrompt);
|
||||
}
|
||||
93
scripts/stubs/obsidian.ts
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
export class App {}
|
||||
|
||||
export class Notice {
|
||||
constructor(public message?: string) {
|
||||
if (message) {
|
||||
console.warn(`[Notice] ${message}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export class TFile {
|
||||
path: string;
|
||||
basename: string;
|
||||
extension: string;
|
||||
|
||||
constructor(path: string) {
|
||||
this.path = path;
|
||||
this.basename = path.split("/").pop() || path;
|
||||
const parts = this.basename.split(".");
|
||||
this.extension = parts.length > 1 ? parts.pop() || "" : "";
|
||||
}
|
||||
}
|
||||
|
||||
export class Vault {
|
||||
getRoot() {
|
||||
return { name: "root" };
|
||||
}
|
||||
|
||||
getAbstractFileByPath() {
|
||||
return null;
|
||||
}
|
||||
|
||||
async read() {
|
||||
return "";
|
||||
}
|
||||
|
||||
getMarkdownFiles() {
|
||||
return [];
|
||||
}
|
||||
|
||||
getAllLoadedFiles() {
|
||||
return [];
|
||||
}
|
||||
|
||||
adapter = {
|
||||
mkdir: async () => {
|
||||
/* no-op */
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export const Platform = {
|
||||
isDesktop: true,
|
||||
isMobile: false,
|
||||
};
|
||||
|
||||
export function normalizePath(path: string): string {
|
||||
return path;
|
||||
}
|
||||
|
||||
export async function requestUrl(): Promise<never> {
|
||||
throw new Error("requestUrl is not available in the CLI environment.");
|
||||
}
|
||||
|
||||
export function getAllTags(): string[] {
|
||||
return [];
|
||||
}
|
||||
|
||||
export class MarkdownView {}
|
||||
|
||||
export class TAbstractFile {}
|
||||
|
||||
export class WorkspaceLeaf {
|
||||
async openFile(): Promise<void> {
|
||||
/* no-op */
|
||||
}
|
||||
}
|
||||
|
||||
export class ItemView {}
|
||||
|
||||
export class Modal {
|
||||
open(): void {
|
||||
/* no-op */
|
||||
}
|
||||
|
||||
close(): void {
|
||||
/* no-op */
|
||||
}
|
||||
}
|
||||
|
||||
export function parseYaml(_: string): unknown {
|
||||
return {};
|
||||
}
|
||||
91
scripts/test-vault.sh
Executable file
|
|
@ -0,0 +1,91 @@
|
|||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
OBSIDIAN_BIN="/Applications/Obsidian.app/Contents/MacOS/obsidian"
|
||||
|
||||
if [[ -z "${COPILOT_TEST_VAULT_PATH:-}" ]]; then
|
||||
cat >&2 <<'EOF'
|
||||
error: COPILOT_TEST_VAULT_PATH is not set.
|
||||
|
||||
Set it once at the user level (e.g. in ~/.zshrc or ~/.config/fish/config.fish)
|
||||
to the absolute path of an Obsidian vault you've opened at least once:
|
||||
|
||||
export COPILOT_TEST_VAULT_PATH="$HOME/Obsidian/CopilotTestVault"
|
||||
|
||||
Then re-run: npm run test:vault
|
||||
EOF
|
||||
exit 1
|
||||
fi
|
||||
|
||||
VAULT_PATH="$COPILOT_TEST_VAULT_PATH"
|
||||
|
||||
if [[ ! -d "$VAULT_PATH" ]]; then
|
||||
echo "error: vault directory not found: $VAULT_PATH" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ ! -d "$VAULT_PATH/.obsidian" ]]; then
|
||||
echo "error: $VAULT_PATH has no .obsidian/ folder." >&2
|
||||
echo "Open the folder as a vault in Obsidian once, then re-run." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
WORKTREE_ROOT="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
cd "$WORKTREE_ROOT"
|
||||
|
||||
echo "==> Installing dependencies"
|
||||
npm install --prefer-offline --no-audit --no-fund
|
||||
|
||||
echo "==> Building plugin"
|
||||
npm run build
|
||||
|
||||
PLUGIN_ID="$(node -p "require('./manifest.json').id")"
|
||||
if [[ -z "$PLUGIN_ID" ]]; then
|
||||
echo "error: could not read plugin id from manifest.json" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
PLUGIN_DIR="$VAULT_PATH/.obsidian/plugins/$PLUGIN_ID"
|
||||
mkdir -p "$PLUGIN_DIR"
|
||||
|
||||
echo "==> Linking artifacts into $PLUGIN_DIR"
|
||||
for f in main.js styles.css; do
|
||||
if [[ ! -f "$WORKTREE_ROOT/$f" ]]; then
|
||||
echo "error: expected build artifact missing: $WORKTREE_ROOT/$f" >&2
|
||||
exit 1
|
||||
fi
|
||||
ln -sfn "$WORKTREE_ROOT/$f" "$PLUGIN_DIR/$f"
|
||||
done
|
||||
|
||||
# Write a branch- and timestamp-tagged manifest.json (real file, not a symlink)
|
||||
# so Obsidian's Community plugins list visibly reflects which worktree/branch
|
||||
# is loaded and when this build was deployed.
|
||||
BRANCH="$(git -C "$WORKTREE_ROOT" rev-parse --abbrev-ref HEAD 2>/dev/null || echo unknown)"
|
||||
BUILD_TS="$(date +%Y%m%d-%H%M%S)"
|
||||
echo "==> Writing branch-tagged manifest.json (branch: $BRANCH, build: $BUILD_TS)"
|
||||
rm -f "$PLUGIN_DIR/manifest.json"
|
||||
SRC="$WORKTREE_ROOT/manifest.json" DEST="$PLUGIN_DIR/manifest.json" BRANCH="$BRANCH" BUILD_TS="$BUILD_TS" node -e '
|
||||
const fs = require("fs");
|
||||
const m = JSON.parse(fs.readFileSync(process.env.SRC, "utf8"));
|
||||
m.name = m.name + " [" + process.env.BRANCH + " @ " + process.env.BUILD_TS + "]";
|
||||
m.description = "[branch: " + process.env.BRANCH + " | build: " + process.env.BUILD_TS + "] " + m.description;
|
||||
fs.writeFileSync(process.env.DEST, JSON.stringify(m, null, 2) + "\n");
|
||||
'
|
||||
|
||||
echo "==> Reloading plugin in Obsidian"
|
||||
if [[ ! -x "$OBSIDIAN_BIN" ]]; then
|
||||
echo "warning: Obsidian CLI not found at $OBSIDIAN_BIN; skipping reload." >&2
|
||||
else
|
||||
if ! "$OBSIDIAN_BIN" plugin:enable id="$PLUGIN_ID" >/dev/null 2>&1 \
|
||||
|| ! "$OBSIDIAN_BIN" plugin:reload id="$PLUGIN_ID" >/dev/null 2>&1; then
|
||||
echo "warning: Obsidian doesn't appear to be running. Start it and the symlinked plugin will load on next open." >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "Done."
|
||||
echo " worktree: $WORKTREE_ROOT"
|
||||
echo " branch: $BRANCH"
|
||||
echo " build: $BUILD_TS"
|
||||
echo " vault: $VAULT_PATH"
|
||||
echo " plugin: $PLUGIN_ID"
|
||||
834
src/LLMProviders/BedrockChatModel.test.ts
Normal file
|
|
@ -0,0 +1,834 @@
|
|||
import { BedrockChatModel } from "./BedrockChatModel";
|
||||
|
||||
type ProcessStreamResult = {
|
||||
deltaChunks: Array<{
|
||||
text?: string;
|
||||
message: {
|
||||
content: unknown;
|
||||
additional_kwargs?: { delta?: { reasoning?: string } };
|
||||
};
|
||||
}>;
|
||||
usage?: Record<string, unknown>;
|
||||
stopReason?: string;
|
||||
hasText: boolean;
|
||||
debugSummaries: string[];
|
||||
};
|
||||
|
||||
type ContentItem = { type: string; text?: string; thinking?: string };
|
||||
|
||||
type ImageContent = {
|
||||
type: "image";
|
||||
source: { type: "base64"; media_type: string; data: string };
|
||||
} | null;
|
||||
|
||||
type RequestBody = {
|
||||
thinking?:
|
||||
| { type: "enabled"; budget_tokens: number }
|
||||
| { type: "adaptive"; display?: "summarized" | "omitted" };
|
||||
temperature?: number;
|
||||
anthropic_version?: string;
|
||||
messages: Array<{
|
||||
role: string;
|
||||
content: Array<{
|
||||
type: string;
|
||||
text?: string;
|
||||
source?: { type: string; media_type: string; data: string };
|
||||
}>;
|
||||
}>;
|
||||
};
|
||||
|
||||
type BedrockInternal = {
|
||||
decodeChunkBytes: (encoded: string) => string[];
|
||||
processStreamEvent: (
|
||||
event: unknown,
|
||||
runManager: unknown,
|
||||
currentUsage: unknown,
|
||||
currentStopReason: unknown
|
||||
) => Promise<ProcessStreamResult>;
|
||||
buildContentItemsFromDelta: (event: unknown) => ContentItem[] | null;
|
||||
extractStreamText: (event: unknown) => string | null;
|
||||
buildRequestBody: (messages: unknown[], options?: unknown) => RequestBody;
|
||||
convertImageContent: (imageUrl: string) => ImageContent;
|
||||
normaliseMessageContent: (
|
||||
message: unknown
|
||||
) => string | Array<{ type: string; [key: string]: unknown }>;
|
||||
};
|
||||
|
||||
const asInternal = (m: BedrockChatModel): BedrockInternal => m as unknown as BedrockInternal;
|
||||
|
||||
/**
|
||||
* Builds a minimal Amazon EventStream message containing the provided UTF-8 payload.
|
||||
* This helper keeps CRC fields at zero because the decoder ignores them.
|
||||
*/
|
||||
const buildEventStreamChunk = (payload: string): string => {
|
||||
const encoder = typeof TextEncoder !== "undefined" ? new TextEncoder() : null;
|
||||
const payloadBytes = encoder ? encoder.encode(payload) : Buffer.from(payload, "utf-8");
|
||||
const headersLength = 0;
|
||||
const totalLength = 12 + headersLength + payloadBytes.length + 4;
|
||||
|
||||
const buffer = new Uint8Array(totalLength);
|
||||
const view = new DataView(buffer.buffer);
|
||||
|
||||
view.setUint32(0, totalLength, false);
|
||||
view.setUint32(4, headersLength, false);
|
||||
view.setUint32(8, 0, false); // Prelude CRC (ignored by decoder)
|
||||
|
||||
buffer.set(payloadBytes, 12);
|
||||
view.setUint32(totalLength - 4, 0, false); // Message CRC (ignored by decoder)
|
||||
|
||||
return Buffer.from(buffer).toString("base64");
|
||||
};
|
||||
|
||||
const createModel = (
|
||||
enableThinking = false,
|
||||
modelId = "anthropic.claude-3-haiku-20240307-v1:0"
|
||||
): BedrockChatModel =>
|
||||
new BedrockChatModel({
|
||||
modelId,
|
||||
apiKey: "test-key",
|
||||
endpoint: `https://example.com/model/${encodeURIComponent(modelId)}/invoke`,
|
||||
streamEndpoint: `https://example.com/model/${encodeURIComponent(modelId)}/invoke-with-response-stream`,
|
||||
anthropicVersion: "bedrock-2023-05-31",
|
||||
enableThinking,
|
||||
fetchImplementation: jest.fn(),
|
||||
});
|
||||
|
||||
const createModelWithFetch = (
|
||||
fetchMock: jest.Mock,
|
||||
opts?: { modelId?: string; noStream?: boolean }
|
||||
): BedrockChatModel =>
|
||||
new BedrockChatModel({
|
||||
modelId: opts?.modelId ?? "anthropic.claude-sonnet-4-5-20250929-v1:0",
|
||||
apiKey: "test-key",
|
||||
endpoint: "https://example.com/model/anthropic.claude-sonnet-4-5-20250929-v1%3A0/invoke",
|
||||
...(opts?.noStream
|
||||
? {}
|
||||
: {
|
||||
streamEndpoint:
|
||||
"https://example.com/model/anthropic.claude-sonnet-4-5-20250929-v1%3A0/invoke-with-response-stream",
|
||||
}),
|
||||
anthropicVersion: "bedrock-2023-05-31",
|
||||
fetchImplementation: fetchMock,
|
||||
});
|
||||
|
||||
describe("BedrockChatModel streaming decode", () => {
|
||||
it("decodes simple base64 JSON payloads", () => {
|
||||
const payload = JSON.stringify({
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: { text: "Hello there" },
|
||||
},
|
||||
});
|
||||
|
||||
const base64 = Buffer.from(payload, "utf-8").toString("base64");
|
||||
|
||||
const model = createModel();
|
||||
const decoded = asInternal(model).decodeChunkBytes(base64);
|
||||
|
||||
expect(decoded).toEqual([payload]);
|
||||
});
|
||||
|
||||
it("extracts payloads from Amazon EventStream encoded chunks", () => {
|
||||
const payload = JSON.stringify({
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: { type: "text_delta", text: "Streaming works!" },
|
||||
},
|
||||
});
|
||||
|
||||
const base64 = buildEventStreamChunk(payload);
|
||||
const model = createModel();
|
||||
const decoded = asInternal(model).decodeChunkBytes(base64);
|
||||
|
||||
expect(decoded).toEqual([payload]);
|
||||
});
|
||||
|
||||
it("produces ChatGenerationChunk entries for decoded deltas", async () => {
|
||||
const payload = JSON.stringify({
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: { type: "text_delta", text: "Chunk text" },
|
||||
},
|
||||
});
|
||||
|
||||
const base64 = buildEventStreamChunk(payload);
|
||||
const event = {
|
||||
type: "chunk",
|
||||
chunk: { bytes: base64 },
|
||||
};
|
||||
|
||||
const model = createModel();
|
||||
const processed = await asInternal(model).processStreamEvent(
|
||||
event,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined
|
||||
);
|
||||
|
||||
expect(processed.hasText).toBe(true);
|
||||
expect(processed.deltaChunks).toHaveLength(1);
|
||||
expect(processed.deltaChunks[0]?.text).toBe("Chunk text");
|
||||
});
|
||||
|
||||
describe("thinking content support", () => {
|
||||
it("buildContentItemsFromDelta recognizes thinking delta type", () => {
|
||||
const event = {
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: {
|
||||
type: "thinking",
|
||||
thinking: "Let me analyze this problem...",
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
const model = createModel();
|
||||
const contentItems = asInternal(model).buildContentItemsFromDelta(event);
|
||||
|
||||
expect(contentItems).toHaveLength(1);
|
||||
expect(contentItems![0]).toEqual({
|
||||
type: "thinking",
|
||||
thinking: "Let me analyze this problem...",
|
||||
});
|
||||
});
|
||||
|
||||
it("buildContentItemsFromDelta recognizes text_delta type", () => {
|
||||
const event = {
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: {
|
||||
type: "text_delta",
|
||||
text: "Based on my analysis, the answer is...",
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
const model = createModel();
|
||||
const contentItems = asInternal(model).buildContentItemsFromDelta(event);
|
||||
|
||||
expect(contentItems).toHaveLength(1);
|
||||
expect(contentItems![0]).toEqual({
|
||||
type: "text",
|
||||
text: "Based on my analysis, the answer is...",
|
||||
});
|
||||
});
|
||||
|
||||
it("processStreamEvent returns chunks with thinking content array", async () => {
|
||||
const payload = JSON.stringify({
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: {
|
||||
type: "thinking",
|
||||
thinking: "Reasoning through this...",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const base64 = buildEventStreamChunk(payload);
|
||||
const event = {
|
||||
type: "chunk",
|
||||
chunk: { bytes: base64 },
|
||||
};
|
||||
|
||||
const model = createModel();
|
||||
const processed = await asInternal(model).processStreamEvent(
|
||||
event,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined
|
||||
);
|
||||
|
||||
expect(processed.hasText).toBe(true);
|
||||
expect(processed.deltaChunks).toHaveLength(1);
|
||||
const chunk = processed.deltaChunks[0];
|
||||
expect(chunk?.text).toBe("Reasoning through this...");
|
||||
|
||||
// Check that content is an array with thinking type
|
||||
expect(Array.isArray(chunk?.message.content)).toBe(true);
|
||||
const content = chunk?.message.content as unknown[];
|
||||
expect(content).toHaveLength(1);
|
||||
expect(content[0]).toEqual({
|
||||
type: "thinking",
|
||||
thinking: "Reasoning through this...",
|
||||
});
|
||||
|
||||
// Check for OpenRouter compatibility
|
||||
expect(chunk?.message.additional_kwargs).toBeDefined();
|
||||
expect(chunk?.message.additional_kwargs?.delta).toEqual({
|
||||
reasoning: "Reasoning through this...",
|
||||
});
|
||||
});
|
||||
|
||||
it("processStreamEvent returns chunks with text content array", async () => {
|
||||
const payload = JSON.stringify({
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: {
|
||||
type: "text_delta",
|
||||
text: "Here is my final answer.",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const base64 = buildEventStreamChunk(payload);
|
||||
const event = {
|
||||
type: "chunk",
|
||||
chunk: { bytes: base64 },
|
||||
};
|
||||
|
||||
const model = createModel();
|
||||
const processed = await asInternal(model).processStreamEvent(
|
||||
event,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined
|
||||
);
|
||||
|
||||
expect(processed.hasText).toBe(true);
|
||||
expect(processed.deltaChunks).toHaveLength(1);
|
||||
const chunk = processed.deltaChunks[0];
|
||||
expect(chunk?.text).toBe("Here is my final answer.");
|
||||
|
||||
// Check that content is an array with text type
|
||||
expect(Array.isArray(chunk?.message.content)).toBe(true);
|
||||
const content = chunk?.message.content as unknown[];
|
||||
expect(content).toHaveLength(1);
|
||||
expect(content[0]).toEqual({
|
||||
type: "text",
|
||||
text: "Here is my final answer.",
|
||||
});
|
||||
|
||||
// No additional_kwargs.delta for regular text
|
||||
expect(chunk?.message.additional_kwargs?.delta).toBeUndefined();
|
||||
});
|
||||
|
||||
it("handles mixed thinking and text deltas correctly", async () => {
|
||||
const model = createModel();
|
||||
|
||||
// First chunk: thinking
|
||||
const thinkingPayload = JSON.stringify({
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: {
|
||||
type: "thinking",
|
||||
thinking: "First, I'll consider...",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const thinkingBase64 = buildEventStreamChunk(thinkingPayload);
|
||||
const thinkingEvent = {
|
||||
type: "chunk",
|
||||
chunk: { bytes: thinkingBase64 },
|
||||
};
|
||||
|
||||
const thinkingResult = await asInternal(model).processStreamEvent(
|
||||
thinkingEvent,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined
|
||||
);
|
||||
|
||||
expect(thinkingResult.deltaChunks).toHaveLength(1);
|
||||
const thinkingChunk = thinkingResult.deltaChunks[0];
|
||||
expect((thinkingChunk?.message.content as Array<{ type: string }>)[0]?.type).toBe("thinking");
|
||||
|
||||
// Second chunk: text
|
||||
const textPayload = JSON.stringify({
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
index: 0,
|
||||
delta: {
|
||||
type: "text_delta",
|
||||
text: "Therefore, the answer is X.",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const textBase64 = buildEventStreamChunk(textPayload);
|
||||
const textEvent = {
|
||||
type: "chunk",
|
||||
chunk: { bytes: textBase64 },
|
||||
};
|
||||
|
||||
const textResult = await asInternal(model).processStreamEvent(
|
||||
textEvent,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined
|
||||
);
|
||||
|
||||
expect(textResult.deltaChunks).toHaveLength(1);
|
||||
const textChunk = textResult.deltaChunks[0];
|
||||
expect((textChunk?.message.content as Array<{ type: string }>)[0]?.type).toBe("text");
|
||||
});
|
||||
|
||||
it("extractStreamText can fallback to extract thinking content", () => {
|
||||
const event = {
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
delta: {
|
||||
type: "thinking",
|
||||
thinking: "Fallback thinking extraction",
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
const model = createModel();
|
||||
const extracted = asInternal(model).extractStreamText(event);
|
||||
|
||||
expect(extracted).toBe("Fallback thinking extraction");
|
||||
});
|
||||
|
||||
it("handles empty thinking content gracefully", () => {
|
||||
const event = {
|
||||
type: "content_block_delta",
|
||||
content_block_delta: {
|
||||
delta: {
|
||||
type: "thinking",
|
||||
thinking: "",
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
const model = createModel();
|
||||
const contentItems = asInternal(model).buildContentItemsFromDelta(event);
|
||||
|
||||
expect(contentItems).toHaveLength(1);
|
||||
expect(contentItems![0]).toEqual({
|
||||
type: "thinking",
|
||||
thinking: "",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("thinking mode enablement", () => {
|
||||
it("includes thinking parameter when enableThinking is true", () => {
|
||||
const model = createModel(true);
|
||||
const requestBody = asInternal(model).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]);
|
||||
|
||||
expect(requestBody.thinking).toEqual({
|
||||
type: "enabled",
|
||||
budget_tokens: 2048,
|
||||
});
|
||||
expect(requestBody.temperature).toBe(1);
|
||||
expect(requestBody.anthropic_version).toBe("bedrock-2023-05-31");
|
||||
});
|
||||
|
||||
it("does not include thinking parameter when enableThinking is false", () => {
|
||||
const model = createModel(false);
|
||||
const requestBody = asInternal(model).buildRequestBody(
|
||||
[{ role: "user", content: "test", getType: () => "human" }],
|
||||
{ temperature: 0.7 }
|
||||
);
|
||||
|
||||
expect(requestBody.thinking).toBeUndefined();
|
||||
expect(requestBody.temperature).toBe(0.7);
|
||||
// anthropic_version should always be present when provided (required for all Bedrock requests)
|
||||
expect(requestBody.anthropic_version).toBe("bedrock-2023-05-31");
|
||||
});
|
||||
|
||||
it("respects user temperature when thinking is disabled", () => {
|
||||
const model = createModel(false);
|
||||
const requestBody = asInternal(model).buildRequestBody(
|
||||
[{ role: "user", content: "test", getType: () => "human" }],
|
||||
{ temperature: 0.5 }
|
||||
);
|
||||
|
||||
expect(requestBody.temperature).toBe(0.5);
|
||||
expect(requestBody.thinking).toBeUndefined();
|
||||
});
|
||||
|
||||
it("forces temperature to 1 when thinking is enabled", () => {
|
||||
const model = createModel(true);
|
||||
const requestBody = asInternal(model).buildRequestBody(
|
||||
[{ role: "user", content: "test", getType: () => "human" }],
|
||||
{ temperature: 0.5 } // User tries to set 0.5, should be overridden to 1
|
||||
);
|
||||
|
||||
expect(requestBody.temperature).toBe(1);
|
||||
expect(requestBody.thinking).toBeDefined();
|
||||
});
|
||||
|
||||
it("uses adaptive thinking with summarized display for claude-opus-4-7", () => {
|
||||
const model = createModel(true, "anthropic.claude-opus-4-7-20260115-v1:0");
|
||||
const requestBody = asInternal(model).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]);
|
||||
|
||||
expect(requestBody.thinking).toEqual({ type: "adaptive", display: "summarized" });
|
||||
expect(requestBody.temperature).toBe(1);
|
||||
});
|
||||
|
||||
it("uses adaptive thinking for opus-4-7 cross-region inference profiles", () => {
|
||||
const model = createModel(true, "global.anthropic.claude-opus-4-7-20260115-v1:0");
|
||||
const requestBody = asInternal(model).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]);
|
||||
|
||||
expect(requestBody.thinking).toEqual({ type: "adaptive", display: "summarized" });
|
||||
});
|
||||
|
||||
it("keeps legacy thinking for opus-4-6 and earlier", () => {
|
||||
const model = createModel(true, "anthropic.claude-opus-4-6-20250115-v1:0");
|
||||
const requestBody = asInternal(model).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]);
|
||||
|
||||
expect(requestBody.thinking).toEqual({ type: "enabled", budget_tokens: 2048 });
|
||||
});
|
||||
|
||||
it("keeps legacy thinking for sonnet-4 and 3-7-sonnet", () => {
|
||||
const sonnet45 = createModel(true, "anthropic.claude-sonnet-4-5-20250929-v1:0");
|
||||
expect(
|
||||
asInternal(sonnet45).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]).thinking
|
||||
).toEqual({ type: "enabled", budget_tokens: 2048 });
|
||||
|
||||
const sonnet37 = createModel(true, "anthropic.claude-3-7-sonnet-20250219-v1:0");
|
||||
expect(
|
||||
asInternal(sonnet37).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]).thinking
|
||||
).toEqual({ type: "enabled", budget_tokens: 2048 });
|
||||
});
|
||||
|
||||
it("keeps legacy thinking for dated Opus 4.0 snapshot IDs", () => {
|
||||
// anthropic.claude-opus-4-20250514-v1:0 is the dated snapshot of Opus 4.0, not 4.20250514.
|
||||
const opus40 = createModel(true, "anthropic.claude-opus-4-20250514-v1:0");
|
||||
expect(
|
||||
asInternal(opus40).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]).thinking
|
||||
).toEqual({ type: "enabled", budget_tokens: 2048 });
|
||||
|
||||
// anthropic.claude-opus-4-1-20250805-v1:0 is dated 4.1, not adaptive.
|
||||
const opus41 = createModel(true, "anthropic.claude-opus-4-1-20250805-v1:0");
|
||||
expect(
|
||||
asInternal(opus41).buildRequestBody([
|
||||
{ role: "user", content: "test", getType: () => "human" },
|
||||
]).thinking
|
||||
).toEqual({ type: "enabled", budget_tokens: 2048 });
|
||||
});
|
||||
});
|
||||
|
||||
describe("vision support", () => {
|
||||
describe("convertImageContent", () => {
|
||||
it("converts valid data URL to Claude image format", () => {
|
||||
const model = createModel();
|
||||
const dataUrl = "data:image/jpeg;base64,/9j/4AAQSkZJRg==";
|
||||
const result = asInternal(model).convertImageContent(dataUrl);
|
||||
|
||||
expect(result).toEqual({
|
||||
type: "image",
|
||||
source: {
|
||||
type: "base64",
|
||||
media_type: "image/jpeg",
|
||||
data: "/9j/4AAQSkZJRg==",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("handles PNG images", () => {
|
||||
const model = createModel();
|
||||
const dataUrl = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUg==";
|
||||
const result = asInternal(model).convertImageContent(dataUrl);
|
||||
|
||||
expect(result).toEqual({
|
||||
type: "image",
|
||||
source: {
|
||||
type: "base64",
|
||||
media_type: "image/png",
|
||||
data: "iVBORw0KGgoAAAANSUhEUg==",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("returns null for invalid data URL format", () => {
|
||||
const model = createModel();
|
||||
const invalidUrl = "not-a-data-url";
|
||||
const result = asInternal(model).convertImageContent(invalidUrl);
|
||||
|
||||
expect(result).toBeNull();
|
||||
});
|
||||
|
||||
it("returns null for non-image media type", () => {
|
||||
const model = createModel();
|
||||
const dataUrl = "data:text/plain;base64,SGVsbG8gV29ybGQ=";
|
||||
const result = asInternal(model).convertImageContent(dataUrl);
|
||||
|
||||
expect(result).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("normaliseMessageContent", () => {
|
||||
it("preserves array content with images", () => {
|
||||
const model = createModel();
|
||||
const message = {
|
||||
content: [
|
||||
{ type: "text", text: "What's in this image?" },
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: "data:image/jpeg;base64,/9j/4AAQSkZJRg==" },
|
||||
},
|
||||
],
|
||||
getType: () => "human",
|
||||
};
|
||||
|
||||
const result = asInternal(model).normaliseMessageContent(message);
|
||||
|
||||
expect(Array.isArray(result)).toBe(true);
|
||||
expect(result).toHaveLength(2);
|
||||
expect(result[0]).toEqual({ type: "text", text: "What's in this image?" });
|
||||
expect(result[1]).toEqual({
|
||||
type: "image_url",
|
||||
image_url: { url: "data:image/jpeg;base64,/9j/4AAQSkZJRg==" },
|
||||
});
|
||||
});
|
||||
|
||||
it("flattens array content without images to string", () => {
|
||||
const model = createModel();
|
||||
const message = {
|
||||
content: [
|
||||
{ type: "text", text: "Hello " },
|
||||
{ type: "text", text: "world!" },
|
||||
],
|
||||
getType: () => "human",
|
||||
};
|
||||
|
||||
const result = asInternal(model).normaliseMessageContent(message);
|
||||
|
||||
expect(typeof result).toBe("string");
|
||||
expect(result).toBe("Hello world!");
|
||||
});
|
||||
|
||||
it("returns string content unchanged", () => {
|
||||
const model = createModel();
|
||||
const message = {
|
||||
content: "Simple text message",
|
||||
getType: () => "human",
|
||||
};
|
||||
|
||||
const result = asInternal(model).normaliseMessageContent(message);
|
||||
|
||||
expect(result).toBe("Simple text message");
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildRequestBody with images", () => {
|
||||
it("includes images in request body for multimodal messages", () => {
|
||||
const model = createModel();
|
||||
const messages = [
|
||||
{
|
||||
content: [
|
||||
{ type: "text", text: "What's in this image?" },
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: "data:image/jpeg;base64,/9j/4AAQSkZJRg==" },
|
||||
},
|
||||
],
|
||||
getType: () => "human",
|
||||
},
|
||||
];
|
||||
|
||||
const requestBody = asInternal(model).buildRequestBody(messages);
|
||||
|
||||
expect(requestBody.messages).toHaveLength(1);
|
||||
expect(requestBody.messages[0].content).toHaveLength(2);
|
||||
|
||||
// Check text block
|
||||
expect(requestBody.messages[0].content[0]).toEqual({
|
||||
type: "text",
|
||||
text: "What's in this image?",
|
||||
});
|
||||
|
||||
// Check image block (converted to Claude format)
|
||||
expect(requestBody.messages[0].content[1]).toEqual({
|
||||
type: "image",
|
||||
source: {
|
||||
type: "base64",
|
||||
media_type: "image/jpeg",
|
||||
data: "/9j/4AAQSkZJRg==",
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("handles multiple images in a single message", () => {
|
||||
const model = createModel();
|
||||
const messages = [
|
||||
{
|
||||
content: [
|
||||
{ type: "text", text: "Compare these images:" },
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: "data:image/jpeg;base64,IMAGE1DATA" },
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: "data:image/png;base64,IMAGE2DATA" },
|
||||
},
|
||||
],
|
||||
getType: () => "human",
|
||||
},
|
||||
];
|
||||
|
||||
const requestBody = asInternal(model).buildRequestBody(messages);
|
||||
|
||||
expect(requestBody.messages[0].content).toHaveLength(3);
|
||||
expect(requestBody.messages[0].content[0].type).toBe("text");
|
||||
expect(requestBody.messages[0].content[1].type).toBe("image");
|
||||
expect(requestBody.messages[0].content[1].source!.media_type).toBe("image/jpeg");
|
||||
expect(requestBody.messages[0].content[2].type).toBe("image");
|
||||
expect(requestBody.messages[0].content[2].source!.media_type).toBe("image/png");
|
||||
});
|
||||
|
||||
it("handles text-only messages correctly", () => {
|
||||
const model = createModel();
|
||||
const messages = [
|
||||
{
|
||||
content: "Just text, no images",
|
||||
getType: () => "human",
|
||||
},
|
||||
];
|
||||
|
||||
const requestBody = asInternal(model).buildRequestBody(messages);
|
||||
|
||||
expect(requestBody.messages).toHaveLength(1);
|
||||
expect(requestBody.messages[0].content).toHaveLength(1);
|
||||
expect(requestBody.messages[0].content[0]).toEqual({
|
||||
type: "text",
|
||||
text: "Just text, no images",
|
||||
});
|
||||
});
|
||||
|
||||
it("skips invalid images and keeps valid content", () => {
|
||||
const model = createModel();
|
||||
const messages = [
|
||||
{
|
||||
content: [
|
||||
{ type: "text", text: "Valid text" },
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: "invalid-url" }, // Invalid - should be skipped
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: { url: "data:image/jpeg;base64,VALIDDATA" }, // Valid
|
||||
},
|
||||
],
|
||||
getType: () => "human",
|
||||
},
|
||||
];
|
||||
|
||||
const requestBody = asInternal(model).buildRequestBody(messages);
|
||||
|
||||
// Should have text + 1 valid image (invalid one skipped)
|
||||
expect(requestBody.messages[0].content).toHaveLength(2);
|
||||
expect(requestBody.messages[0].content[0].type).toBe("text");
|
||||
expect(requestBody.messages[0].content[1].type).toBe("image");
|
||||
});
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("BedrockChatModel inference-profile error rewriting", () => {
|
||||
const awsInferenceProfileError = JSON.stringify({
|
||||
message:
|
||||
"Invocation of model ID anthropic.claude-sonnet-4-5 with on-demand throughput isn't supported. Retry your request with the ID or ARN of an inference profile that contains this model.",
|
||||
});
|
||||
|
||||
const makeErrorResponse = (status: number, body: string): Response =>
|
||||
({
|
||||
ok: false,
|
||||
status,
|
||||
text: () => Promise.resolve(body),
|
||||
}) as unknown as Response;
|
||||
|
||||
it("rewrites 400 inference-profile error in non-streaming path to actionable message", async () => {
|
||||
const fetchMock = jest.fn().mockResolvedValue(makeErrorResponse(400, awsInferenceProfileError));
|
||||
const model = createModelWithFetch(fetchMock, { noStream: true });
|
||||
|
||||
const messages = [{ content: "hi", getType: () => "human", type: "human" }];
|
||||
await expect(model._generate(messages as never, {})).rejects.toThrow(
|
||||
/cross-region inference profile ID/
|
||||
);
|
||||
});
|
||||
|
||||
it("rewrites 400 inference-profile error in streaming path to actionable message", async () => {
|
||||
const fetchMock = jest.fn().mockResolvedValue(makeErrorResponse(400, awsInferenceProfileError));
|
||||
const model = createModelWithFetch(fetchMock);
|
||||
|
||||
const messages = [{ content: "hi", getType: () => "human", type: "human" }];
|
||||
const gen = model._streamResponseChunks(messages as never, {});
|
||||
await expect(gen.next()).rejects.toThrow(/cross-region inference profile ID/);
|
||||
});
|
||||
|
||||
it("includes the bare model ID in the rewritten message", async () => {
|
||||
const fetchMock = jest.fn().mockResolvedValue(makeErrorResponse(400, awsInferenceProfileError));
|
||||
const model = createModelWithFetch(fetchMock, { noStream: true });
|
||||
|
||||
const messages = [{ content: "hi", getType: () => "human", type: "human" }];
|
||||
await expect(model._generate(messages as never, {})).rejects.toThrow(
|
||||
/anthropic\.claude-sonnet-4-5/
|
||||
);
|
||||
});
|
||||
|
||||
it("does not rewrite a 400 error that is unrelated to inference profiles", async () => {
|
||||
const genericBody = JSON.stringify({ message: "ValidationException: bad request" });
|
||||
const fetchMock = jest.fn().mockResolvedValue(makeErrorResponse(400, genericBody));
|
||||
const model = createModelWithFetch(fetchMock, { noStream: true });
|
||||
|
||||
const messages = [{ content: "hi", getType: () => "human", type: "human" }];
|
||||
await expect(model._generate(messages as never, {})).rejects.toThrow(
|
||||
/Amazon Bedrock request failed with status 400/
|
||||
);
|
||||
});
|
||||
|
||||
it("does not rewrite non-400 errors", async () => {
|
||||
const body = JSON.stringify({ message: "Internal Server Error" });
|
||||
const fetchMock = jest.fn().mockResolvedValue(makeErrorResponse(500, body));
|
||||
const model = createModelWithFetch(fetchMock, { noStream: true });
|
||||
|
||||
const messages = [{ content: "hi", getType: () => "human", type: "human" }];
|
||||
await expect(model._generate(messages as never, {})).rejects.toThrow(
|
||||
/Amazon Bedrock request failed with status 500/
|
||||
);
|
||||
});
|
||||
|
||||
it("rewrites the error even when AWS uses a curly apostrophe in 'isn’t supported'", async () => {
|
||||
const curlyApostropheBody = JSON.stringify({
|
||||
message:
|
||||
"Invocation of model ID anthropic.claude-sonnet-4-5 with on-demand throughput isn’t supported. Retry your request with the ID or ARN of an inference profile that contains this model.",
|
||||
});
|
||||
const fetchMock = jest.fn().mockResolvedValue(makeErrorResponse(400, curlyApostropheBody));
|
||||
const model = createModelWithFetch(fetchMock, { noStream: true });
|
||||
|
||||
const messages = [{ content: "hi", getType: () => "human", type: "human" }];
|
||||
await expect(model._generate(messages as never, {})).rejects.toThrow(
|
||||
/cross-region inference profile ID/
|
||||
);
|
||||
});
|
||||
|
||||
it("uses the provider segment from the bare model ID in the prefix guidance", async () => {
|
||||
const nonAnthropicBody = JSON.stringify({
|
||||
message:
|
||||
"Invocation of model ID meta.llama4-maverick-17b with on-demand throughput isn't supported. Retry your request with the ID or ARN of an inference profile that contains this model.",
|
||||
});
|
||||
const fetchMock = jest.fn().mockResolvedValue(makeErrorResponse(400, nonAnthropicBody));
|
||||
const model = createModelWithFetch(fetchMock, { noStream: true });
|
||||
|
||||
const messages = [{ content: "hi", getType: () => "human", type: "human" }];
|
||||
await expect(model._generate(messages as never, {})).rejects.toThrow(/global\.meta\.<id>/);
|
||||
});
|
||||
});
|
||||
1617
src/LLMProviders/BedrockChatModel.ts
Normal file
87
src/LLMProviders/ChatLMStudio.ts
Normal file
|
|
@ -0,0 +1,87 @@
|
|||
import { ChatOpenAI } from "@langchain/openai";
|
||||
|
||||
/**
|
||||
* ChatLMStudio extends ChatOpenAI with the Responses API (/v1/responses)
|
||||
* for LM Studio local inference.
|
||||
*
|
||||
* Patches LangChain/OpenAI SDK compatibility issues with LM Studio:
|
||||
* - Ensures text.format is always set (LM Studio requires it)
|
||||
* - Removes strict:null from tool definitions (LM Studio rejects it)
|
||||
*/
|
||||
export interface ChatLMStudioInput {
|
||||
modelName?: string;
|
||||
apiKey?: string;
|
||||
configuration?: Record<string, unknown>;
|
||||
temperature?: number;
|
||||
maxTokens?: number;
|
||||
topP?: number;
|
||||
frequencyPenalty?: number;
|
||||
streaming?: boolean;
|
||||
streamUsage?: boolean;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a fetch wrapper that sanitizes request bodies for LM Studio
|
||||
* compatibility. This intercepts at the HTTP level, which is the last
|
||||
* stop before the request is sent, guaranteeing all null values in
|
||||
* tools are stripped regardless of which LangChain code path produced them.
|
||||
*/
|
||||
function createLMStudioFetch(baseFetch?: typeof window.fetch): typeof window.fetch {
|
||||
const underlyingFetch = baseFetch || window.fetch;
|
||||
|
||||
return async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
if (init?.body && typeof init.body === "string") {
|
||||
try {
|
||||
const body = JSON.parse(init.body) as { tools?: unknown };
|
||||
let modified = false;
|
||||
|
||||
// Strip null/undefined values from tool definitions
|
||||
if (Array.isArray(body.tools)) {
|
||||
body.tools = body.tools.map((tool: Record<string, unknown>) => {
|
||||
const cleaned: Record<string, unknown> = {};
|
||||
for (const [key, value] of Object.entries(tool)) {
|
||||
if (value !== null && value !== undefined) {
|
||||
cleaned[key] = value;
|
||||
}
|
||||
}
|
||||
return cleaned;
|
||||
});
|
||||
modified = true;
|
||||
}
|
||||
|
||||
if (modified) {
|
||||
init = { ...init, body: JSON.stringify(body) };
|
||||
}
|
||||
} catch {
|
||||
// Not JSON, pass through unchanged
|
||||
}
|
||||
}
|
||||
return underlyingFetch(input, init);
|
||||
};
|
||||
}
|
||||
|
||||
export class ChatLMStudio extends ChatOpenAI {
|
||||
constructor(fields: ChatLMStudioInput) {
|
||||
const configuration = fields.configuration as { fetch?: typeof window.fetch } | undefined;
|
||||
const originalFetch = configuration?.fetch;
|
||||
|
||||
super({
|
||||
...fields,
|
||||
useResponsesApi: true,
|
||||
configuration: {
|
||||
...fields.configuration,
|
||||
// Wrap fetch to sanitize request bodies for LM Studio compatibility
|
||||
fetch: createLMStudioFetch(originalFetch),
|
||||
},
|
||||
// modelKwargs is spread LAST in ChatOpenAIResponses.invocationParams(),
|
||||
// overriding the computed `text` field. Without this, LangChain emits
|
||||
// `text: { format: undefined }` (serializes to `text: {}`) which LM Studio
|
||||
// rejects with "Required: text.format".
|
||||
modelKwargs: {
|
||||
...(fields.modelKwargs as Record<string, unknown> | undefined),
|
||||
text: { format: { type: "text" } },
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
560
src/LLMProviders/ChatOpenRouter.ts
Normal file
|
|
@ -0,0 +1,560 @@
|
|||
import { BaseChatModelParams } from "@langchain/core/language_models/chat_models";
|
||||
import { AIMessageChunk, BaseMessage } from "@langchain/core/messages";
|
||||
import type { UsageMetadata } from "@langchain/core/messages";
|
||||
import { ChatGenerationChunk } from "@langchain/core/outputs";
|
||||
import { ChatOpenAI } from "@langchain/openai";
|
||||
import OpenAI from "openai";
|
||||
import { logInfo } from "@/logger";
|
||||
|
||||
type OpenRouterChatChunk = OpenAI.ChatCompletionChunk;
|
||||
type OpenRouterUsage = NonNullable<OpenRouterChatChunk["usage"]>;
|
||||
type OpenRouterMessageParam = OpenAI.ChatCompletionMessageParam;
|
||||
|
||||
/**
|
||||
* ChatOpenRouter extends ChatOpenAI to support OpenRouter-specific features,
|
||||
* particularly reasoning/thinking tokens.
|
||||
*
|
||||
* OpenRouter exposes thinking tokens via the `reasoning` request parameter
|
||||
* and responds with `reasoning_details` in both streaming and non-streaming modes.
|
||||
*
|
||||
* @see https://openrouter.ai/docs/use-cases/reasoning-tokens
|
||||
*/
|
||||
export interface ChatOpenRouterInput extends BaseChatModelParams {
|
||||
/**
|
||||
* Enable reasoning/thinking tokens from OpenRouter
|
||||
* When true, requests will include reasoning parameters
|
||||
*/
|
||||
enableReasoning?: boolean;
|
||||
|
||||
/**
|
||||
* Reasoning effort level: "minimal", "low", "medium", "high", or "xhigh"
|
||||
* Controls the amount of reasoning the model uses
|
||||
* Note: "minimal" will be treated as "low" for OpenRouter
|
||||
* Note: "xhigh" is only supported by GPT-5.4 models
|
||||
*/
|
||||
reasoningEffort?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
|
||||
/**
|
||||
* Enable prompt caching (cache_control) for OpenRouter requests.
|
||||
* Defaults to true. Set to false for Zero Data Retention (ZDR) endpoints
|
||||
* that do not support prompt caching.
|
||||
*/
|
||||
enablePromptCaching?: boolean;
|
||||
|
||||
// All other ChatOpenAI parameters
|
||||
modelName?: string;
|
||||
apiKey?: string;
|
||||
configuration?: {
|
||||
baseURL?: string;
|
||||
defaultHeaders?: Record<string, string>;
|
||||
fetch?: (input: RequestInfo | URL, init?: RequestInit) => Promise<Response>;
|
||||
[key: string]: unknown;
|
||||
};
|
||||
temperature?: number;
|
||||
maxTokens?: number;
|
||||
topP?: number;
|
||||
frequencyPenalty?: number;
|
||||
streaming?: boolean;
|
||||
maxRetries?: number;
|
||||
maxConcurrency?: number;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
export class ChatOpenRouter extends ChatOpenAI {
|
||||
private enableReasoning: boolean;
|
||||
private reasoningEffort?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
private enablePromptCaching: boolean;
|
||||
private openaiClient: OpenAI;
|
||||
/** True when the configured baseURL belongs to the OpenRouter gateway. */
|
||||
private isOpenRouter: boolean;
|
||||
|
||||
constructor(fields: ChatOpenRouterInput) {
|
||||
const {
|
||||
enableReasoning = false,
|
||||
reasoningEffort,
|
||||
enablePromptCaching = true,
|
||||
...rest
|
||||
} = fields;
|
||||
|
||||
// Pass all other parameters to ChatOpenAI
|
||||
super(rest);
|
||||
|
||||
this.enableReasoning = enableReasoning;
|
||||
this.reasoningEffort = reasoningEffort;
|
||||
this.enablePromptCaching = enablePromptCaching;
|
||||
|
||||
const baseURL = fields.configuration?.baseURL || "https://openrouter.ai/api/v1";
|
||||
this.isOpenRouter = baseURL.includes("openrouter.ai");
|
||||
|
||||
// Create our own OpenAI client for raw access
|
||||
this.openaiClient = new OpenAI({
|
||||
apiKey: fields.apiKey,
|
||||
baseURL,
|
||||
defaultHeaders: fields.configuration?.defaultHeaders,
|
||||
fetch: fields.configuration?.fetch,
|
||||
dangerouslyAllowBrowser: true,
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Override the invocation parameters to include prompt caching and reasoning when enabled.
|
||||
*
|
||||
* Prompt caching: The `cache_control` field opts Anthropic models (via OpenRouter) into
|
||||
* automatic cache breakpoint detection, reducing token costs on repeated context. This
|
||||
* field is only sent when connected to the OpenRouter gateway; other backends (LM Studio,
|
||||
* Copilot Plus) use the same class but must not receive OpenRouter-specific fields.
|
||||
*
|
||||
* @see https://openrouter.ai/docs/features/prompt-caching
|
||||
*/
|
||||
override invocationParams(options?: this["ParsedCallOptions"]): Record<string, unknown> {
|
||||
const baseParams = super.invocationParams(options);
|
||||
|
||||
// Only inject cache_control for OpenRouter endpoints. LM Studio, Copilot Plus, and
|
||||
// other OpenAI-compatible backends share this class but reject unknown top-level fields.
|
||||
// Skip caching when enablePromptCaching is false (e.g. for ZDR endpoints that don't
|
||||
// support Anthropic's automatic caching).
|
||||
const withCaching =
|
||||
this.isOpenRouter && this.enablePromptCaching
|
||||
? { ...baseParams, cache_control: { type: "ephemeral" } }
|
||||
: baseParams;
|
||||
|
||||
// Add reasoning parameter if enabled
|
||||
if (this.enableReasoning) {
|
||||
// Per OpenRouter docs:
|
||||
// - For Anthropic models: MUST use reasoning.max_tokens or reasoning.effort
|
||||
// - For other models: Can use reasoning.enabled
|
||||
// - max_tokens must be strictly higher than reasoning budget
|
||||
|
||||
// Prefer effort if provided, otherwise fall back to max_tokens
|
||||
if (this.reasoningEffort) {
|
||||
// Map "minimal" to "low" since OpenRouter doesn't support "minimal"
|
||||
const effort = this.reasoningEffort === "minimal" ? "low" : this.reasoningEffort;
|
||||
logInfo(`OpenRouter reasoning enabled with effort: ${effort}`);
|
||||
return {
|
||||
...withCaching,
|
||||
reasoning: {
|
||||
effort,
|
||||
},
|
||||
};
|
||||
} else {
|
||||
logInfo(`OpenRouter reasoning enabled with max_tokens: 1024`);
|
||||
return {
|
||||
...withCaching,
|
||||
reasoning: {
|
||||
max_tokens: 1024,
|
||||
},
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return withCaching;
|
||||
}
|
||||
|
||||
/**
|
||||
* Override to use raw OpenAI SDK to access reasoning_details
|
||||
* LangChain filters out reasoning_details, so we bypass it completely
|
||||
*/
|
||||
override async *_streamResponseChunks(
|
||||
messages: BaseMessage[],
|
||||
options: this["ParsedCallOptions"],
|
||||
_runManager?: { handleLLMNewToken: (token: string) => Promise<void> }
|
||||
): AsyncGenerator<ChatGenerationChunk> {
|
||||
const params = this.invocationParams(options);
|
||||
const openaiMessages = this.toOpenRouterMessages(messages);
|
||||
|
||||
const stream = (await this.openaiClient.chat.completions.create({
|
||||
...params,
|
||||
messages: openaiMessages,
|
||||
stream: true,
|
||||
stream_options: {
|
||||
...(params.stream_options ?? {}),
|
||||
include_usage: true,
|
||||
},
|
||||
} as Parameters<
|
||||
typeof this.openaiClient.chat.completions.create
|
||||
>[0])) as unknown as AsyncIterable<OpenRouterChatChunk>;
|
||||
|
||||
let usageSummary: OpenRouterUsage | undefined;
|
||||
|
||||
for await (const rawChunk of stream) {
|
||||
if (rawChunk.usage) {
|
||||
usageSummary = rawChunk.usage;
|
||||
}
|
||||
|
||||
const choice = rawChunk.choices?.[0];
|
||||
const delta = choice?.delta;
|
||||
if (!choice || !delta) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const reasoningText = this.normalizeReasoningChunk(
|
||||
(delta as Record<string, unknown>)?.reasoning
|
||||
);
|
||||
const reasoningDetails = this.extractReasoningDetails(choice);
|
||||
const content = this.extractDeltaContent(delta.content);
|
||||
|
||||
const messageChunk = this.buildMessageChunk({
|
||||
rawChunk,
|
||||
delta: delta as unknown as Record<string, unknown>,
|
||||
content,
|
||||
finishReason: choice.finish_reason,
|
||||
reasoningDetails,
|
||||
reasoningText,
|
||||
});
|
||||
|
||||
const generationChunk = new ChatGenerationChunk({
|
||||
message: messageChunk,
|
||||
text: typeof messageChunk.content === "string" ? messageChunk.content : "",
|
||||
generationInfo: {
|
||||
finish_reason: choice.finish_reason,
|
||||
// Reason: system_fingerprint is marked deprecated by some scorecards but is still
|
||||
// returned by OpenAI-style streaming APIs and is useful for telemetry.
|
||||
system_fingerprint: rawChunk["system_fingerprint"],
|
||||
model: rawChunk.model,
|
||||
},
|
||||
});
|
||||
|
||||
yield generationChunk;
|
||||
if (generationChunk.text) {
|
||||
await _runManager?.handleLLMNewToken(generationChunk.text);
|
||||
}
|
||||
}
|
||||
|
||||
if (usageSummary) {
|
||||
yield this.buildUsageGenerationChunk(usageSummary);
|
||||
}
|
||||
|
||||
if (options.signal?.aborted) {
|
||||
throw new Error("AbortError");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert LangChain messages to OpenRouter-ready messages.
|
||||
*
|
||||
* @param messages LangChain messages passed into the model
|
||||
* @returns Messages formatted for the OpenRouter API
|
||||
*/
|
||||
private toOpenRouterMessages(messages: BaseMessage[]): OpenRouterMessageParam[] {
|
||||
return messages.map((msg) => {
|
||||
const msgRecord = msg as unknown as Record<string, unknown>;
|
||||
const role =
|
||||
typeof msg._getType === "function"
|
||||
? msg._getType()
|
||||
: ((msgRecord.role as string) ?? "user");
|
||||
const mappedRole =
|
||||
role === "human"
|
||||
? "user"
|
||||
: role === "ai"
|
||||
? "assistant"
|
||||
: (role as OpenAI.ChatCompletionRole);
|
||||
|
||||
if (msgRecord.tool_call_id) {
|
||||
return {
|
||||
role: "tool",
|
||||
content: msg.content,
|
||||
tool_call_id: msgRecord.tool_call_id as string,
|
||||
} as OpenRouterMessageParam;
|
||||
}
|
||||
|
||||
if (msg.additional_kwargs?.function_call) {
|
||||
return {
|
||||
role: mappedRole,
|
||||
content: msg.content,
|
||||
function_call: msg.additional_kwargs.function_call,
|
||||
} as OpenRouterMessageParam;
|
||||
}
|
||||
|
||||
// Handle modern tool_calls format (used by autonomous agent)
|
||||
if (msg.additional_kwargs?.tool_calls) {
|
||||
return {
|
||||
role: mappedRole,
|
||||
content: msg.content,
|
||||
tool_calls: msg.additional_kwargs.tool_calls,
|
||||
} as OpenRouterMessageParam;
|
||||
}
|
||||
|
||||
return {
|
||||
role: mappedRole,
|
||||
content: msg.content,
|
||||
} as OpenRouterMessageParam;
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Build an `AIMessageChunk` enriched with reasoning metadata.
|
||||
*
|
||||
* @param config Chunk configuration values extracted from the stream
|
||||
* @returns AI message chunk ready for downstream streaming utilities
|
||||
*/
|
||||
private buildMessageChunk(config: {
|
||||
rawChunk: OpenRouterChatChunk;
|
||||
delta: Record<string, unknown>;
|
||||
content: string;
|
||||
finishReason: string | null | undefined;
|
||||
reasoningText?: string;
|
||||
reasoningDetails?: unknown[];
|
||||
}): AIMessageChunk {
|
||||
const { rawChunk, delta, content, finishReason, reasoningText, reasoningDetails } = config;
|
||||
const toolCallChunks = this.extractToolCallChunks(delta.tool_calls);
|
||||
|
||||
const additionalKwargs: Record<string, unknown> = {};
|
||||
|
||||
if (delta.function_call) {
|
||||
additionalKwargs.function_call = delta.function_call;
|
||||
}
|
||||
|
||||
if (Array.isArray(delta.tool_calls)) {
|
||||
additionalKwargs.tool_calls = delta.tool_calls;
|
||||
}
|
||||
|
||||
const deltaPayload: Record<string, unknown> = {};
|
||||
if (reasoningText) {
|
||||
deltaPayload.reasoning = reasoningText;
|
||||
}
|
||||
if (reasoningDetails && reasoningDetails.length > 0) {
|
||||
deltaPayload.reasoning_details = reasoningDetails;
|
||||
}
|
||||
|
||||
if (Object.keys(deltaPayload).length > 0) {
|
||||
additionalKwargs.delta = {
|
||||
...(additionalKwargs.delta as Record<string, unknown>),
|
||||
...deltaPayload,
|
||||
};
|
||||
}
|
||||
|
||||
if (reasoningDetails && reasoningDetails.length > 0) {
|
||||
additionalKwargs.reasoning_details = reasoningDetails;
|
||||
}
|
||||
|
||||
const responseMetadata = this.buildResponseMetadata(rawChunk, finishReason);
|
||||
|
||||
return new AIMessageChunk({
|
||||
content,
|
||||
additional_kwargs: additionalKwargs,
|
||||
tool_call_chunks: toolCallChunks,
|
||||
response_metadata: responseMetadata,
|
||||
id: rawChunk.id,
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize streamed reasoning payloads into plain text for the UI.
|
||||
*
|
||||
* @param reasoning Arbitrary reasoning payload returned by OpenRouter
|
||||
* @returns Normalized reasoning text or undefined
|
||||
*/
|
||||
private normalizeReasoningChunk(reasoning: unknown): string | undefined {
|
||||
if (!reasoning) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
if (typeof reasoning === "string") {
|
||||
return reasoning;
|
||||
}
|
||||
|
||||
if (Array.isArray(reasoning)) {
|
||||
return reasoning
|
||||
.map((item) => this.normalizeReasoningChunk(item))
|
||||
.filter((item): item is string => Boolean(item))
|
||||
.join("");
|
||||
}
|
||||
|
||||
if (typeof reasoning === "object") {
|
||||
const record = reasoning as Record<string, unknown>;
|
||||
const candidates = [
|
||||
record.output_text,
|
||||
record.text,
|
||||
record.reasoning,
|
||||
record.thinking,
|
||||
record.content,
|
||||
];
|
||||
|
||||
const normalized = candidates.find((value) => typeof value === "string");
|
||||
if (typeof normalized === "string") {
|
||||
return normalized;
|
||||
}
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract reasoning details arrays from the streamed choice payload.
|
||||
*
|
||||
* @param choice Chunk choice object from the OpenRouter stream
|
||||
* @returns Array of reasoning detail entries, if present
|
||||
*/
|
||||
private extractReasoningDetails(
|
||||
choice: OpenAI.ChatCompletionChunk.Choice
|
||||
): unknown[] | undefined {
|
||||
const choiceRecord = choice as unknown as Record<string, Record<string, unknown>>;
|
||||
const candidate =
|
||||
choiceRecord?.delta?.reasoning_details ??
|
||||
choiceRecord?.message?.reasoning_details ??
|
||||
(choice as unknown as Record<string, unknown>)?.reasoning_details;
|
||||
|
||||
if (!Array.isArray(candidate)) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return (candidate as unknown[]).filter((detail) => detail !== undefined && detail !== null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Flatten OpenRouter delta content into a single text string.
|
||||
*
|
||||
* @param content Delta content payload
|
||||
* @returns Text representation for downstream streaming
|
||||
*/
|
||||
private extractDeltaContent(content: unknown): string {
|
||||
if (typeof content === "string") {
|
||||
return content;
|
||||
}
|
||||
|
||||
if (Array.isArray(content)) {
|
||||
return content
|
||||
.map((part) => {
|
||||
if (typeof part === "string") {
|
||||
return part;
|
||||
}
|
||||
if (
|
||||
part &&
|
||||
typeof part === "object" &&
|
||||
typeof (part as { text?: unknown }).text === "string"
|
||||
) {
|
||||
return (part as { text: string }).text;
|
||||
}
|
||||
return "";
|
||||
})
|
||||
.join("");
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
/**
|
||||
* Map raw OpenRouter tool call deltas into LangChain tool call chunks.
|
||||
*
|
||||
* @param toolCalls Tool call deltas returned by OpenRouter
|
||||
* @returns Tool call chunk array compatible with LangChain
|
||||
*/
|
||||
private extractToolCallChunks(
|
||||
toolCalls: unknown
|
||||
):
|
||||
| Array<{ name?: string; args?: string; id?: string; index?: number; type: "tool_call_chunk" }>
|
||||
| undefined {
|
||||
if (!Array.isArray(toolCalls)) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
return toolCalls.map((rawCall) => {
|
||||
const call = rawCall as
|
||||
| { function?: { name?: string; arguments?: string }; id?: string; index?: number }
|
||||
| null
|
||||
| undefined;
|
||||
return {
|
||||
name: call?.function?.name,
|
||||
args: call?.function?.arguments,
|
||||
id: call?.id,
|
||||
index: call?.index,
|
||||
type: "tool_call_chunk" as const,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Build response metadata payload with finish reason and usage info.
|
||||
*
|
||||
* @param rawChunk Raw streaming chunk from OpenRouter
|
||||
* @param finishReason Stop reason reported by the model
|
||||
* @returns Metadata object attached to each AI message chunk
|
||||
*/
|
||||
private buildResponseMetadata(
|
||||
rawChunk: OpenRouterChatChunk,
|
||||
finishReason: string | null | undefined
|
||||
): Record<string, unknown> {
|
||||
const metadata: Record<string, unknown> = {
|
||||
model_provider: "openrouter",
|
||||
};
|
||||
|
||||
if (finishReason) {
|
||||
metadata.finish_reason = finishReason;
|
||||
}
|
||||
|
||||
if (rawChunk.model) {
|
||||
metadata.model = rawChunk.model;
|
||||
}
|
||||
|
||||
// Reason: system_fingerprint is marked deprecated by some scorecards but is still
|
||||
// returned by OpenAI-style streaming APIs and is useful for telemetry. Use bracket
|
||||
// access to bypass JSDoc deprecation warnings.
|
||||
const fingerprint = rawChunk["system_fingerprint"];
|
||||
if (fingerprint) {
|
||||
metadata.system_fingerprint = fingerprint;
|
||||
}
|
||||
|
||||
if (rawChunk.usage) {
|
||||
metadata.usage = { ...rawChunk.usage };
|
||||
metadata.tokenUsage = {
|
||||
promptTokens: rawChunk.usage.prompt_tokens,
|
||||
completionTokens: rawChunk.usage.completion_tokens,
|
||||
totalTokens: rawChunk.usage.total_tokens,
|
||||
};
|
||||
}
|
||||
|
||||
return metadata;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a terminal usage chunk so downstream consumers can capture token usage.
|
||||
*
|
||||
* @param usage Usage payload returned by the streaming API
|
||||
* @returns Chat generation chunk containing usage metadata
|
||||
*/
|
||||
private buildUsageGenerationChunk(usage: OpenRouterUsage): ChatGenerationChunk {
|
||||
const inputTokenDetails: Record<string, number> = {};
|
||||
const outputTokenDetails: Record<string, number> = {};
|
||||
|
||||
const promptDetails = usage.prompt_tokens_details ?? {};
|
||||
if (typeof promptDetails.audio_tokens === "number") {
|
||||
inputTokenDetails.audio = promptDetails.audio_tokens;
|
||||
}
|
||||
if (typeof promptDetails.cached_tokens === "number") {
|
||||
inputTokenDetails.cache_read = promptDetails.cached_tokens;
|
||||
}
|
||||
|
||||
const completionDetails = usage.completion_tokens_details ?? {};
|
||||
if (typeof completionDetails.audio_tokens === "number") {
|
||||
outputTokenDetails.audio = completionDetails.audio_tokens;
|
||||
}
|
||||
if (typeof completionDetails.reasoning_tokens === "number") {
|
||||
outputTokenDetails.reasoning = completionDetails.reasoning_tokens;
|
||||
}
|
||||
|
||||
const usageMetadata: UsageMetadata = {
|
||||
input_tokens: usage.prompt_tokens ?? 0,
|
||||
output_tokens: usage.completion_tokens ?? 0,
|
||||
total_tokens: usage.total_tokens ?? 0,
|
||||
};
|
||||
|
||||
if (Object.keys(inputTokenDetails).length > 0) {
|
||||
usageMetadata.input_token_details = inputTokenDetails;
|
||||
}
|
||||
|
||||
if (Object.keys(outputTokenDetails).length > 0) {
|
||||
usageMetadata.output_token_details = outputTokenDetails;
|
||||
}
|
||||
|
||||
const messageChunk = new AIMessageChunk({
|
||||
content: "",
|
||||
response_metadata: { usage: { ...usage } },
|
||||
usage_metadata: usageMetadata,
|
||||
});
|
||||
|
||||
return new ChatGenerationChunk({
|
||||
message: messageChunk,
|
||||
text: "",
|
||||
});
|
||||
}
|
||||
}
|
||||
|
|
@ -1,15 +1,186 @@
|
|||
import { JinaEmbeddings, JinaEmbeddingsParams } from "@langchain/community/embeddings/jina";
|
||||
/*
|
||||
* Adapted from @langchain/community JinaEmbeddings.
|
||||
* Copyright (c) LangChain, Inc. Licensed under the MIT License.
|
||||
* Source: https://github.com/langchain-ai/langchainjs-community/blob/886df5749a926f59e6fdf38a3465c62ec9e7ce32/libs/community/src/embeddings/jina.ts
|
||||
*/
|
||||
|
||||
export class CustomJinaEmbeddings extends JinaEmbeddings {
|
||||
import { Embeddings, type EmbeddingsParams } from "@langchain/core/embeddings";
|
||||
import { chunkArray } from "@langchain/core/utils/chunk_array";
|
||||
import { getEnvironmentVariable } from "@langchain/core/utils/env";
|
||||
|
||||
export interface JinaEmbeddingsParams extends EmbeddingsParams {
|
||||
/** Model name to use. */
|
||||
model: string;
|
||||
/** Compatibility alias used by this plugin's embedding manager. */
|
||||
modelName?: string;
|
||||
/** Jina-compatible embeddings endpoint. */
|
||||
baseUrl?: string;
|
||||
/** Timeout to use when making requests to Jina. */
|
||||
timeout?: number;
|
||||
/** The maximum number of documents to embed in a single request. */
|
||||
batchSize?: number;
|
||||
/** Whether to strip new lines from the input text. */
|
||||
stripNewLines?: boolean;
|
||||
/** The dimensions of the embedding. */
|
||||
dimensions?: number;
|
||||
/** Whether to L2-normalize the embedding vectors. */
|
||||
normalized?: boolean;
|
||||
}
|
||||
|
||||
type JinaMultiModelInput =
|
||||
| {
|
||||
text: string;
|
||||
image?: never;
|
||||
}
|
||||
| {
|
||||
image: string;
|
||||
text?: never;
|
||||
};
|
||||
|
||||
export type JinaEmbeddingsInput = string | JinaMultiModelInput;
|
||||
|
||||
interface EmbeddingCreateParams {
|
||||
model: JinaEmbeddingsParams["model"];
|
||||
input: JinaEmbeddingsInput[];
|
||||
dimensions: number;
|
||||
task: "retrieval.query" | "retrieval.passage";
|
||||
normalized?: boolean;
|
||||
}
|
||||
|
||||
interface EmbeddingResponse {
|
||||
model: string;
|
||||
object: string;
|
||||
usage: {
|
||||
total_tokens: number;
|
||||
prompt_tokens: number;
|
||||
};
|
||||
data: {
|
||||
object: string;
|
||||
index: number;
|
||||
embedding: number[];
|
||||
}[];
|
||||
}
|
||||
|
||||
interface EmbeddingErrorResponse {
|
||||
detail: string;
|
||||
}
|
||||
|
||||
export class CustomJinaEmbeddings extends Embeddings implements JinaEmbeddingsParams {
|
||||
model: JinaEmbeddingsParams["model"] = "jina-clip-v2";
|
||||
batchSize = 24;
|
||||
baseUrl = "https://api.jina.ai/v1/embeddings";
|
||||
stripNewLines = true;
|
||||
dimensions = 1024;
|
||||
apiKey: string;
|
||||
normalized = true;
|
||||
|
||||
/**
|
||||
* Creates a Jina embeddings client using local configuration or Jina environment variables.
|
||||
*/
|
||||
constructor(
|
||||
fields?: Partial<JinaEmbeddingsParams> & {
|
||||
apiKey?: string;
|
||||
baseUrl?: string;
|
||||
}
|
||||
) {
|
||||
super(fields);
|
||||
if (fields?.baseUrl) {
|
||||
this.baseUrl = fields.baseUrl;
|
||||
const fieldsWithDefaults = { maxConcurrency: 2, ...fields };
|
||||
super(fieldsWithDefaults);
|
||||
|
||||
const apiKey =
|
||||
fieldsWithDefaults?.apiKey ||
|
||||
getEnvironmentVariable("JINA_API_KEY") ||
|
||||
getEnvironmentVariable("JINA_AUTH_TOKEN");
|
||||
|
||||
if (!apiKey) throw new Error("Jina API key not found");
|
||||
|
||||
this.apiKey = apiKey;
|
||||
this.model = fieldsWithDefaults?.model ?? fieldsWithDefaults?.modelName ?? this.model;
|
||||
this.baseUrl = fieldsWithDefaults?.baseUrl ?? this.baseUrl;
|
||||
this.dimensions = fieldsWithDefaults?.dimensions ?? this.dimensions;
|
||||
this.batchSize = fieldsWithDefaults?.batchSize ?? this.batchSize;
|
||||
this.stripNewLines = fieldsWithDefaults?.stripNewLines ?? this.stripNewLines;
|
||||
this.normalized = fieldsWithDefaults?.normalized ?? this.normalized;
|
||||
}
|
||||
|
||||
/**
|
||||
* Embeds passage documents with Jina retrieval-passage task parameters.
|
||||
*/
|
||||
async embedDocuments(input: JinaEmbeddingsInput[]): Promise<number[][]> {
|
||||
const batches = chunkArray(this.doStripNewLines(input), this.batchSize);
|
||||
const batchRequests = batches.map((batch) => {
|
||||
const params = this.getParams(batch);
|
||||
return this.embeddingWithRetry(params);
|
||||
});
|
||||
|
||||
const batchResponses = await Promise.all(batchRequests);
|
||||
const embeddings: number[][] = [];
|
||||
|
||||
for (let i = 0; i < batchResponses.length; i += 1) {
|
||||
const batch = batches[i];
|
||||
const batchResponse = batchResponses[i] || [];
|
||||
for (let j = 0; j < batch.length; j += 1) {
|
||||
embeddings.push(batchResponse[j]);
|
||||
}
|
||||
}
|
||||
|
||||
return embeddings;
|
||||
}
|
||||
|
||||
/**
|
||||
* Embeds a query with Jina retrieval-query task parameters.
|
||||
*/
|
||||
async embedQuery(input: JinaEmbeddingsInput): Promise<number[]> {
|
||||
const params = this.getParams(this.doStripNewLines([input]), true);
|
||||
const embeddings = (await this.embeddingWithRetry(params)) || [[]];
|
||||
return embeddings[0];
|
||||
}
|
||||
|
||||
/**
|
||||
* Removes newlines from string inputs when configured to match upstream Jina behavior.
|
||||
*/
|
||||
private doStripNewLines(input: JinaEmbeddingsInput[]): JinaEmbeddingsInput[] {
|
||||
if (this.stripNewLines) {
|
||||
return input.map((item) => {
|
||||
if (typeof item === "string") {
|
||||
return item.replace(/\n/g, " ");
|
||||
}
|
||||
if (item.text) {
|
||||
return { text: item.text.replace(/\n/g, " ") };
|
||||
}
|
||||
return item;
|
||||
});
|
||||
}
|
||||
return input;
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds the request body for Jina's retrieval embedding API.
|
||||
*/
|
||||
private getParams(input: JinaEmbeddingsInput[], query?: boolean): EmbeddingCreateParams {
|
||||
return {
|
||||
model: this.model,
|
||||
input,
|
||||
dimensions: this.dimensions,
|
||||
task: query ? "retrieval.query" : "retrieval.passage",
|
||||
normalized: this.normalized,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Sends a single embeddings request and returns vectors in response order.
|
||||
*/
|
||||
private async embeddingWithRetry(body: EmbeddingCreateParams): Promise<number[][]> {
|
||||
const response = await fetch(this.baseUrl, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${this.apiKey}`,
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
});
|
||||
const embeddingData: EmbeddingResponse | EmbeddingErrorResponse = await response.json();
|
||||
if ("detail" in embeddingData && embeddingData.detail) {
|
||||
throw new Error(`${embeddingData.detail}`);
|
||||
}
|
||||
return (embeddingData as EmbeddingResponse).data.map(({ embedding }) => embedding);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
71
src/LLMProviders/CustomOpenAIEmbeddings.ts
Normal file
|
|
@ -0,0 +1,71 @@
|
|||
import { safeFetchNoThrow } from "@/utils";
|
||||
import { OpenAIEmbeddings } from "@langchain/openai";
|
||||
|
||||
export class CustomOpenAIEmbeddings extends OpenAIEmbeddings {
|
||||
private customConfig: Record<string, unknown>;
|
||||
|
||||
constructor(config: Record<string, unknown>) {
|
||||
super(config);
|
||||
// Store the config for our custom methods
|
||||
this.customConfig = config;
|
||||
}
|
||||
|
||||
async embedQuery(text: string): Promise<number[]> {
|
||||
// Make direct API call to avoid OpenAI client's response processing
|
||||
const embedding = await this.callEmbeddingAPI([text]);
|
||||
return embedding[0];
|
||||
}
|
||||
|
||||
async embedDocuments(texts: string[]): Promise<number[][]> {
|
||||
// Make direct API call to avoid OpenAI client's response processing
|
||||
const embeddings = await this.callEmbeddingAPI(texts);
|
||||
return embeddings;
|
||||
}
|
||||
|
||||
private async callEmbeddingAPI(texts: string[]): Promise<number[][]> {
|
||||
const requestBody = {
|
||||
model: this.customConfig.modelName,
|
||||
input: texts,
|
||||
encoding_format: "float",
|
||||
};
|
||||
|
||||
// Get the correct baseURL, apiKey, and fetch function from the configuration
|
||||
const configuration = this.customConfig.configuration as
|
||||
| { baseURL?: string; fetch?: typeof fetch }
|
||||
| undefined;
|
||||
const baseURL = configuration?.baseURL || "https://api.openai.com/v1";
|
||||
const url = `${baseURL}/embeddings`;
|
||||
const apiKey = this.customConfig.apiKey as string;
|
||||
const fetchFn = configuration?.fetch || safeFetchNoThrow;
|
||||
|
||||
const response = await fetchFn(url, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
"Content-Type": "application/json",
|
||||
...((this.customConfig.headers as Record<string, string>) || {}),
|
||||
},
|
||||
body: JSON.stringify(requestBody),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
throw new Error(
|
||||
`Embedding API request failed: ${response.status} ${response.statusText} - ${errorText}`
|
||||
);
|
||||
}
|
||||
|
||||
const responseData = (await response.json()) as { data?: Array<{ embedding?: unknown }> };
|
||||
|
||||
if (!responseData.data || !Array.isArray(responseData.data)) {
|
||||
throw new Error("Invalid API response format: missing or invalid data array");
|
||||
}
|
||||
|
||||
return responseData.data.map((item) => {
|
||||
if (!item.embedding || !Array.isArray(item.embedding)) {
|
||||
throw new Error("Invalid API response format: missing or invalid embedding array");
|
||||
}
|
||||
return item.embedding as number[];
|
||||
});
|
||||
}
|
||||
}
|
||||
|
|
@ -1,23 +1,88 @@
|
|||
import { BREVILABS_API_BASE_URL } from "@/constants";
|
||||
import { getDecryptedKey } from "@/encryptionService";
|
||||
import { MissingPlusLicenseError } from "@/error";
|
||||
import { logInfo } from "@/logger";
|
||||
import { turnOffPlus, turnOnPlus } from "@/plusUtils";
|
||||
import { getSettings } from "@/settings/model";
|
||||
import { Buffer } from "buffer";
|
||||
import { Notice } from "obsidian";
|
||||
import { arrayBufferToBase64 } from "@/utils/base64";
|
||||
import { requestUrl } from "obsidian";
|
||||
|
||||
export interface BrocaResponse {
|
||||
response: {
|
||||
tool_calls: Array<{
|
||||
tool: string;
|
||||
args: {
|
||||
[key: string]: any;
|
||||
};
|
||||
}>;
|
||||
salience_terms: string[];
|
||||
/**
|
||||
* Build a multipart/form-data body buffer from a FormData instance.
|
||||
* Returned as an ArrayBuffer suitable for passing to Obsidian's requestUrl.
|
||||
*
|
||||
* @param formData - FormData containing strings and/or File/Blob entries.
|
||||
* @returns The serialized multipart body and the Content-Type header (including boundary).
|
||||
*/
|
||||
async function buildMultipartFromFormData(
|
||||
formData: FormData
|
||||
): Promise<{ body: ArrayBuffer; contentType: string }> {
|
||||
const boundary = `----CopilotBoundary${Math.random().toString(16).slice(2)}${Date.now().toString(16)}`;
|
||||
const encoder = new TextEncoder();
|
||||
const parts: Uint8Array[] = [];
|
||||
|
||||
for (const [name, value] of formData.entries()) {
|
||||
parts.push(encoder.encode(`--${boundary}\r\n`));
|
||||
if (value instanceof Blob) {
|
||||
const filename = value instanceof File ? value.name : "blob";
|
||||
const contentType = value.type || "application/octet-stream";
|
||||
parts.push(
|
||||
encoder.encode(
|
||||
`Content-Disposition: form-data; name="${name}"; filename="${filename}"\r\n` +
|
||||
`Content-Type: ${contentType}\r\n\r\n`
|
||||
)
|
||||
);
|
||||
const buf = await value.arrayBuffer();
|
||||
parts.push(new Uint8Array(buf));
|
||||
parts.push(encoder.encode("\r\n"));
|
||||
} else {
|
||||
parts.push(encoder.encode(`Content-Disposition: form-data; name="${name}"\r\n\r\n`));
|
||||
parts.push(encoder.encode(String(value)));
|
||||
parts.push(encoder.encode("\r\n"));
|
||||
}
|
||||
}
|
||||
parts.push(encoder.encode(`--${boundary}--\r\n`));
|
||||
|
||||
const totalLength = parts.reduce((sum, p) => sum + p.byteLength, 0);
|
||||
const out = new Uint8Array(totalLength);
|
||||
let offset = 0;
|
||||
for (const part of parts) {
|
||||
out.set(part, offset);
|
||||
offset += part.byteLength;
|
||||
}
|
||||
return {
|
||||
body: out.buffer.slice(out.byteOffset, out.byteOffset + out.byteLength),
|
||||
contentType: `multipart/form-data; boundary=${boundary}`,
|
||||
};
|
||||
elapsed_time_ms: number;
|
||||
detail?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize a requestUrl response into the {data, error} shape used by Brevilabs API methods.
|
||||
* Handles the case where `response.json` is a raw string (non-JSON body, e.g. HTML error page).
|
||||
*/
|
||||
function parseBrevilabsResponse<T>(
|
||||
response: { status: number; json: unknown },
|
||||
endpoint: string
|
||||
): { data: T | null; error?: Error } {
|
||||
let data: unknown = response.json;
|
||||
if (typeof data === "string") {
|
||||
try {
|
||||
data = JSON.parse(data);
|
||||
} catch {
|
||||
// Non-JSON body — fall through to status-based error.
|
||||
}
|
||||
}
|
||||
if (response.status < 200 || response.status >= 300) {
|
||||
const detail = (data as { detail?: { reason?: string; error?: string } } | null)?.detail;
|
||||
if (detail?.reason) {
|
||||
const error = new Error(detail.reason);
|
||||
if (detail.error) error.name = detail.error;
|
||||
return { data: null, error };
|
||||
}
|
||||
return { data: null, error: new Error(`HTTP error: ${response.status}`) };
|
||||
}
|
||||
logInfo(`[API ${endpoint} request]:`, data);
|
||||
return { data: data as T };
|
||||
}
|
||||
|
||||
export interface RerankResponse {
|
||||
|
|
@ -36,22 +101,22 @@ export interface RerankResponse {
|
|||
}
|
||||
|
||||
export interface ToolCall {
|
||||
tool: any;
|
||||
args: any;
|
||||
tool: unknown;
|
||||
args: unknown;
|
||||
}
|
||||
|
||||
export interface Url4llmResponse {
|
||||
response: any;
|
||||
response: string;
|
||||
elapsed_time_ms: number;
|
||||
}
|
||||
|
||||
export interface Pdf4llmResponse {
|
||||
response: any;
|
||||
response: string;
|
||||
elapsed_time_ms: number;
|
||||
}
|
||||
|
||||
export interface Docs4llmResponse {
|
||||
response: any;
|
||||
response: unknown;
|
||||
elapsed_time_ms: number;
|
||||
}
|
||||
|
||||
|
|
@ -76,25 +141,16 @@ export interface Youtube4llmResponse {
|
|||
elapsed_time_ms: number;
|
||||
}
|
||||
|
||||
export interface Twitter4llmResponse {
|
||||
response: string;
|
||||
elapsed_time_ms: number;
|
||||
}
|
||||
|
||||
export interface LicenseResponse {
|
||||
is_valid: boolean;
|
||||
plan: string;
|
||||
}
|
||||
|
||||
export interface AutocompleteResponse {
|
||||
response: {
|
||||
completion: string;
|
||||
};
|
||||
elapsed_time_ms: number;
|
||||
}
|
||||
|
||||
export interface WordCompleteResponse {
|
||||
response: {
|
||||
selected_word: string;
|
||||
};
|
||||
elapsed_time_ms: number;
|
||||
}
|
||||
|
||||
export class BrevilabsClient {
|
||||
private static instance: BrevilabsClient;
|
||||
private pluginVersion: string = "Unknown";
|
||||
|
|
@ -108,10 +164,9 @@ export class BrevilabsClient {
|
|||
|
||||
private checkLicenseKey() {
|
||||
if (!getSettings().plusLicenseKey) {
|
||||
new Notice(
|
||||
throw new MissingPlusLicenseError(
|
||||
"Copilot Plus license key not found. Please enter your license key in the settings."
|
||||
);
|
||||
throw new Error("License key not initialized");
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -121,7 +176,7 @@ export class BrevilabsClient {
|
|||
|
||||
private async makeRequest<T>(
|
||||
endpoint: string,
|
||||
body: any,
|
||||
body: Record<string, unknown>,
|
||||
method = "POST",
|
||||
excludeAuthHeader = false,
|
||||
skipLicenseCheck = false
|
||||
|
|
@ -139,32 +194,21 @@ export class BrevilabsClient {
|
|||
url.searchParams.append(key, value as string);
|
||||
});
|
||||
}
|
||||
|
||||
const response = await fetch(url.toString(), {
|
||||
method,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
...(!excludeAuthHeader && {
|
||||
Authorization: `Bearer ${await getDecryptedKey(getSettings().plusLicenseKey)}`,
|
||||
}),
|
||||
"X-Client-Version": this.pluginVersion,
|
||||
},
|
||||
...(method === "POST" && { body: JSON.stringify(body) }),
|
||||
});
|
||||
const data = await response.json();
|
||||
if (!response.ok) {
|
||||
try {
|
||||
const errorDetail = data.detail;
|
||||
const error = new Error(errorDetail.reason);
|
||||
error.name = errorDetail.error;
|
||||
return { data: null, error };
|
||||
} catch {
|
||||
return { data: null, error: new Error("Unknown error") };
|
||||
}
|
||||
const headers: Record<string, string> = {
|
||||
"Content-Type": "application/json",
|
||||
"X-Client-Version": this.pluginVersion,
|
||||
};
|
||||
if (!excludeAuthHeader) {
|
||||
headers.Authorization = `Bearer ${await getDecryptedKey(getSettings().plusLicenseKey)}`;
|
||||
}
|
||||
logInfo(`==== ${endpoint} request ====:`, data);
|
||||
|
||||
return { data };
|
||||
const response = await requestUrl({
|
||||
url: url.toString(),
|
||||
method,
|
||||
headers,
|
||||
...(method === "POST" && { body: JSON.stringify(body) }),
|
||||
throw: false,
|
||||
});
|
||||
return parseBrevilabsResponse<T>(response, endpoint);
|
||||
}
|
||||
|
||||
private async makeFormDataRequest<T>(
|
||||
|
|
@ -182,29 +226,21 @@ export class BrevilabsClient {
|
|||
const url = new URL(`${BREVILABS_API_BASE_URL}${endpoint}`);
|
||||
|
||||
try {
|
||||
const response = await fetch(url.toString(), {
|
||||
// Build multipart body manually for requestUrl (does not natively support FormData).
|
||||
const { body, contentType } = await buildMultipartFromFormData(formData);
|
||||
|
||||
const response = await requestUrl({
|
||||
url: url.toString(),
|
||||
method: "POST",
|
||||
headers: {
|
||||
// No Content-Type header - browser will set it automatically with boundary
|
||||
"Content-Type": contentType,
|
||||
Authorization: `Bearer ${await getDecryptedKey(getSettings().plusLicenseKey)}`,
|
||||
"X-Client-Version": this.pluginVersion,
|
||||
},
|
||||
body: formData,
|
||||
body,
|
||||
throw: false,
|
||||
});
|
||||
|
||||
const data = await response.json();
|
||||
if (!response.ok) {
|
||||
try {
|
||||
const errorDetail = data.detail;
|
||||
const error = new Error(errorDetail.reason);
|
||||
error.name = errorDetail.error;
|
||||
return { data: null, error };
|
||||
} catch {
|
||||
return { data: null, error: new Error(`HTTP error: ${response.status}`) };
|
||||
}
|
||||
}
|
||||
logInfo(`==== ${endpoint} FormData request ====:`, data);
|
||||
return { data };
|
||||
return parseBrevilabsResponse<T>(response, `${endpoint} form-data`);
|
||||
} catch (error) {
|
||||
return { data: null, error: error instanceof Error ? error : new Error(String(error)) };
|
||||
}
|
||||
|
|
@ -212,19 +248,45 @@ export class BrevilabsClient {
|
|||
|
||||
/**
|
||||
* Validate the license key and update the isPlusUser setting.
|
||||
* @param context Optional context object containing the features that the user is using to validate the license key.
|
||||
* @returns true if the license key is valid, false if the license key is invalid, and undefined if
|
||||
* unknown error.
|
||||
*/
|
||||
async validateLicenseKey(): Promise<{ isValid: boolean | undefined; plan?: string }> {
|
||||
async validateLicenseKey(
|
||||
context?: Record<string, unknown>
|
||||
): Promise<{ isValid: boolean | undefined; plan?: string }> {
|
||||
// Build the request body with proper structure
|
||||
const requestBody: Record<string, unknown> = {
|
||||
license_key: await getDecryptedKey(getSettings().plusLicenseKey),
|
||||
};
|
||||
|
||||
// Safely spread context if provided, ensuring no conflicts with required fields
|
||||
if (context && typeof context === "object") {
|
||||
// Filter out any undefined or null values from context
|
||||
const filteredContext = Object.fromEntries(
|
||||
Object.entries(context).filter(([_, value]) => value !== undefined && value !== null)
|
||||
);
|
||||
|
||||
// Remove any reserved fields that must not be overridden by context
|
||||
const reservedKeys = new Set(["license_key", "user_id"]);
|
||||
for (const key of reservedKeys) {
|
||||
if (key in filteredContext) {
|
||||
delete (filteredContext as Record<string, unknown>)[key];
|
||||
}
|
||||
}
|
||||
|
||||
// Spread the filtered context into the request body
|
||||
Object.assign(requestBody, filteredContext);
|
||||
}
|
||||
|
||||
const { data, error } = await this.makeRequest<LicenseResponse>(
|
||||
"/license",
|
||||
{
|
||||
license_key: await getDecryptedKey(getSettings().plusLicenseKey),
|
||||
},
|
||||
requestBody,
|
||||
"POST",
|
||||
true,
|
||||
true
|
||||
);
|
||||
|
||||
if (error) {
|
||||
if (error.message === "Invalid license key") {
|
||||
turnOffPlus();
|
||||
|
|
@ -237,21 +299,6 @@ export class BrevilabsClient {
|
|||
return { isValid: true, plan: data?.plan };
|
||||
}
|
||||
|
||||
async broca(userMessage: string, isProjectMode: boolean): Promise<BrocaResponse> {
|
||||
const { data, error } = await this.makeRequest<BrocaResponse>("/broca", {
|
||||
message: userMessage,
|
||||
is_project_mode: isProjectMode,
|
||||
});
|
||||
if (error) {
|
||||
throw error;
|
||||
}
|
||||
if (!data) {
|
||||
throw new Error("No data returned from broca");
|
||||
}
|
||||
|
||||
return data;
|
||||
}
|
||||
|
||||
async rerank(query: string, documents: string[]): Promise<RerankResponse> {
|
||||
const { data, error } = await this.makeRequest<RerankResponse>("/rerank", {
|
||||
query,
|
||||
|
|
@ -282,7 +329,7 @@ export class BrevilabsClient {
|
|||
|
||||
async pdf4llm(binaryContent: ArrayBuffer): Promise<Pdf4llmResponse> {
|
||||
// Convert ArrayBuffer to base64 string
|
||||
const base64Content = Buffer.from(binaryContent).toString("base64");
|
||||
const base64Content = arrayBufferToBase64(binaryContent);
|
||||
|
||||
const { data, error } = await this.makeRequest<Pdf4llmResponse>("/pdf4llm", {
|
||||
pdf: base64Content,
|
||||
|
|
@ -392,42 +439,15 @@ export class BrevilabsClient {
|
|||
return data;
|
||||
}
|
||||
|
||||
async autocomplete(
|
||||
prefix: string,
|
||||
noteContext: string = "",
|
||||
relevant_notes: string = ""
|
||||
): Promise<AutocompleteResponse> {
|
||||
const { data, error } = await this.makeRequest<AutocompleteResponse>("/autocomplete", {
|
||||
prompt: prefix,
|
||||
note_context: noteContext,
|
||||
relevant_notes: relevant_notes,
|
||||
max_tokens: 64,
|
||||
});
|
||||
async twitter4llm(url: string): Promise<Twitter4llmResponse> {
|
||||
const { data, error } = await this.makeRequest<Twitter4llmResponse>("/twitter4llm", { url });
|
||||
if (error) {
|
||||
throw error;
|
||||
}
|
||||
if (!data) {
|
||||
throw new Error("No data returned from autocomplete");
|
||||
throw new Error("No data returned from twitter4llm");
|
||||
}
|
||||
return data;
|
||||
}
|
||||
|
||||
async wordcomplete(
|
||||
prefix: string,
|
||||
suffix: string = "",
|
||||
suggestions: string[]
|
||||
): Promise<WordCompleteResponse> {
|
||||
const { data, error } = await this.makeRequest<WordCompleteResponse>("/wordcomplete", {
|
||||
prefix: prefix,
|
||||
suffix: suffix,
|
||||
suggestions: suggestions,
|
||||
});
|
||||
if (error) {
|
||||
throw error;
|
||||
}
|
||||
if (!data) {
|
||||
throw new Error("No data returned from wordcomplete");
|
||||
}
|
||||
return data;
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,41 +1,33 @@
|
|||
import {
|
||||
getChainType,
|
||||
getCurrentProject,
|
||||
getModelKey,
|
||||
SetChainOptions,
|
||||
setChainType,
|
||||
} from "@/aiParams";
|
||||
import ChainFactory, { ChainType, Document } from "@/chainFactory";
|
||||
import { getChainType, getCurrentProject, getModelKey, SetChainOptions } from "@/aiParams";
|
||||
import { ChainType } from "@/chainType";
|
||||
import { BUILTIN_CHAT_MODELS, USER_SENDER } from "@/constants";
|
||||
import {
|
||||
AutonomousAgentChainRunner,
|
||||
ChainRunner,
|
||||
CopilotPlusChainRunner,
|
||||
LLMChainRunner,
|
||||
ProjectChainRunner,
|
||||
VaultQAChainRunner,
|
||||
} from "@/LLMProviders/chainRunner";
|
||||
} from "@/LLMProviders/chainRunner/index";
|
||||
import { logError, logInfo } from "@/logger";
|
||||
import { HybridRetriever } from "@/search/hybridRetriever";
|
||||
import VectorStoreManager from "@/search/vectorStoreManager";
|
||||
import { getSettings, getSystemPrompt, subscribeToSettingsChange } from "@/settings/model";
|
||||
import { ChatMessage } from "@/sharedState";
|
||||
import { findCustomModel, isOSeriesModel, isSupportedChain } from "@/utils";
|
||||
import { getSettings, subscribeToSettingsChange } from "@/settings/model";
|
||||
import { getSystemPrompt } from "@/system-prompts/systemPromptBuilder";
|
||||
import { ChatMessage } from "@/types/message";
|
||||
import { findCustomModel, isOSeriesModel } from "@/utils";
|
||||
import { MissingModelKeyError } from "@/error";
|
||||
import {
|
||||
ChatPromptTemplate,
|
||||
HumanMessagePromptTemplate,
|
||||
MessagesPlaceholder,
|
||||
} from "@langchain/core/prompts";
|
||||
import { RunnableSequence } from "@langchain/core/runnables";
|
||||
import { Document } from "@langchain/core/documents";
|
||||
import { App, Notice } from "obsidian";
|
||||
import ChatModelManager from "./chatModelManager";
|
||||
import MemoryManager from "./memoryManager";
|
||||
import PromptManager from "./promptManager";
|
||||
import { UserMemoryManager } from "@/memory/UserMemoryManager";
|
||||
|
||||
export default class ChainManager {
|
||||
// TODO: These chains are deprecated since we now use direct chat model calls in chain runners
|
||||
// Consider removing after verifying no dependencies remain
|
||||
private chain: RunnableSequence;
|
||||
private retrievalChain: RunnableSequence;
|
||||
private retrievedDocuments: Document[] = [];
|
||||
|
||||
public getRetrievedDocuments(): Document[] {
|
||||
|
|
@ -43,30 +35,27 @@ export default class ChainManager {
|
|||
}
|
||||
|
||||
public app: App;
|
||||
public vectorStoreManager: VectorStoreManager;
|
||||
public chatModelManager: ChatModelManager;
|
||||
public memoryManager: MemoryManager;
|
||||
public promptManager: PromptManager;
|
||||
public userMemoryManager: UserMemoryManager;
|
||||
private pendingModelError: Error | null = null;
|
||||
|
||||
// A chat history that stores the messages sent and received
|
||||
// Only reset when the user explicitly clicks "New Chat"
|
||||
private chatMessages: ChatMessage[] = [];
|
||||
|
||||
constructor(app: App, vectorStoreManager: VectorStoreManager) {
|
||||
this.chatMessages = [];
|
||||
|
||||
constructor(app: App) {
|
||||
// Instantiate singletons
|
||||
this.app = app;
|
||||
this.vectorStoreManager = vectorStoreManager;
|
||||
this.memoryManager = MemoryManager.getInstance();
|
||||
this.chatModelManager = ChatModelManager.getInstance();
|
||||
this.promptManager = PromptManager.getInstance();
|
||||
this.userMemoryManager = new UserMemoryManager(app);
|
||||
|
||||
// Initialize async operations
|
||||
this.initialize();
|
||||
void this.initialize().catch((err) => logError("ChainManager initialize failed", err));
|
||||
|
||||
subscribeToSettingsChange(async () => {
|
||||
await this.createChainWithNewModel();
|
||||
subscribeToSettingsChange(() => {
|
||||
void this.createChainWithNewModel().catch((err) =>
|
||||
logError("createChainWithNewModel failed", err)
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
|
|
@ -74,36 +63,19 @@ export default class ChainManager {
|
|||
await this.createChainWithNewModel();
|
||||
}
|
||||
|
||||
// TODO: These methods are deprecated - chain runners now use direct chat model calls
|
||||
// Remove after confirming no usage remains
|
||||
public getChain(): RunnableSequence {
|
||||
return this.chain;
|
||||
}
|
||||
|
||||
public getRetrievalChain(): RunnableSequence {
|
||||
return this.retrievalChain;
|
||||
}
|
||||
|
||||
private validateChainType(chainType: ChainType): void {
|
||||
if (chainType === undefined || chainType === null) throw new Error("No chain type set");
|
||||
}
|
||||
|
||||
private validateChatModel() {
|
||||
if (this.pendingModelError) {
|
||||
throw this.pendingModelError;
|
||||
}
|
||||
|
||||
if (!this.chatModelManager.validateChatModel(this.chatModelManager.getChatModel())) {
|
||||
const errorMsg =
|
||||
"Chat model is not initialized properly, check your API key in Copilot setting and make sure you have API access.";
|
||||
new Notice(errorMsg);
|
||||
throw new Error(errorMsg);
|
||||
}
|
||||
}
|
||||
|
||||
// TODO: This method is deprecated - chain validation no longer needed
|
||||
// Remove after confirming no dependencies
|
||||
private validateChainInitialization() {
|
||||
if (!this.chain || !isSupportedChain(this.chain)) {
|
||||
console.error("Chain is not initialized properly, re-initializing chain: ", getChainType());
|
||||
this.createChainWithNewModel({}, false);
|
||||
// this.setChain(getChainType());
|
||||
throw new MissingModelKeyError(errorMsg);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -119,6 +91,7 @@ export default class ChainManager {
|
|||
options: SetChainOptions = {},
|
||||
neededReInitChatMode: boolean = true
|
||||
): Promise<void> {
|
||||
let newModelKey: string | undefined;
|
||||
const chainType = getChainType();
|
||||
const currentProject = getCurrentProject();
|
||||
|
||||
|
|
@ -126,15 +99,14 @@ export default class ChainManager {
|
|||
return;
|
||||
}
|
||||
|
||||
let newModelKey =
|
||||
chainType === ChainType.PROJECT_CHAIN ? currentProject?.projectModelKey : getModelKey();
|
||||
|
||||
if (!newModelKey) {
|
||||
new Notice("No model key found");
|
||||
throw new Error("No model key found");
|
||||
}
|
||||
|
||||
try {
|
||||
newModelKey =
|
||||
chainType === ChainType.PROJECT_CHAIN ? currentProject?.projectModelKey : getModelKey();
|
||||
|
||||
if (!newModelKey) {
|
||||
throw new MissingModelKeyError("No model key found. Please select a model in settings.");
|
||||
}
|
||||
|
||||
if (neededReInitChatMode) {
|
||||
let customModel = findCustomModel(newModelKey, getSettings().activeModels);
|
||||
if (!customModel) {
|
||||
|
|
@ -168,131 +140,65 @@ export default class ChainManager {
|
|||
...currentProject?.modelConfigs,
|
||||
};
|
||||
await this.chatModelManager.setChatModel(mergedModel);
|
||||
this.pendingModelError = null;
|
||||
}
|
||||
|
||||
// Must update the chatModel for chain because ChainFactory always
|
||||
// retrieves the old chain without the chatModel change if it exists!
|
||||
// Create a new chain with the new chatModel
|
||||
this.setChain(chainType, options);
|
||||
// Chain-type housekeeping. Do NOT write `chainType` back to the atom —
|
||||
// the atom is owned by the UI dropdowns and `applyPlusSettings`. The
|
||||
// captured local `chainType` may already be stale by the time we reach
|
||||
// here (we just awaited `setChatModel(...)`), and writing it back used
|
||||
// to create a self-sustaining `setChainType` → ProjectManager
|
||||
// subscriber → `createChainWithNewModel` loop that froze Obsidian on
|
||||
// apply-Plus-key.
|
||||
if (this.chatModelManager.validateChatModel(this.chatModelManager.getChatModel())) {
|
||||
this.validateChainType(chainType);
|
||||
if (options.refreshIndex) {
|
||||
await this.refreshVaultIndex();
|
||||
}
|
||||
} else {
|
||||
console.error(
|
||||
"createChainWithNewModel: skipping chain-type housekeeping — no chat model set."
|
||||
);
|
||||
}
|
||||
logInfo(`Setting model to ${newModelKey}`);
|
||||
} catch (error) {
|
||||
this.pendingModelError = error instanceof Error ? error : new Error(String(error));
|
||||
logError(`createChainWithNewModel failed: ${error}`);
|
||||
logInfo(`modelKey: ${newModelKey}`);
|
||||
}
|
||||
}
|
||||
|
||||
// TODO: This method is deprecated - chain runners now handle chain logic directly
|
||||
// Remove after confirming no usage remains
|
||||
async setChain(chainType: ChainType, options: SetChainOptions = {}): Promise<void> {
|
||||
if (!this.chatModelManager.validateChatModel(this.chatModelManager.getChatModel())) {
|
||||
console.error("setChain failed: No chat model set.");
|
||||
return;
|
||||
}
|
||||
|
||||
this.validateChainType(chainType);
|
||||
|
||||
// Get chatModel, memory, prompt, and embeddingAPI from respective managers
|
||||
const chatModel = this.chatModelManager.getChatModel();
|
||||
const memory = this.memoryManager.getMemory();
|
||||
const chatPrompt = this.promptManager.getChatPrompt();
|
||||
|
||||
switch (chainType) {
|
||||
case ChainType.LLM_CHAIN: {
|
||||
// TODO: LLMChainRunner now handles this directly without chains
|
||||
this.chain = ChainFactory.createNewLLMChain({
|
||||
llm: chatModel,
|
||||
memory: memory,
|
||||
prompt: options.prompt || chatPrompt,
|
||||
abortController: options.abortController,
|
||||
}) as RunnableSequence;
|
||||
|
||||
setChainType(ChainType.LLM_CHAIN);
|
||||
break;
|
||||
}
|
||||
|
||||
case ChainType.VAULT_QA_CHAIN: {
|
||||
// TODO: VaultQAChainRunner now handles this directly without chains
|
||||
await this.initializeQAChain(options);
|
||||
|
||||
const retriever = new HybridRetriever({
|
||||
minSimilarityScore: 0.01,
|
||||
maxK: getSettings().maxSourceChunks,
|
||||
salientTerms: [],
|
||||
});
|
||||
|
||||
// Create new conversational retrieval chain
|
||||
this.retrievalChain = ChainFactory.createConversationalRetrievalChain(
|
||||
{
|
||||
llm: chatModel,
|
||||
retriever: retriever,
|
||||
systemMessage: getSystemPrompt(),
|
||||
},
|
||||
this.storeRetrieverDocuments.bind(this),
|
||||
getSettings().debug
|
||||
);
|
||||
|
||||
setChainType(ChainType.VAULT_QA_CHAIN);
|
||||
if (getSettings().debug) {
|
||||
console.log("New Vault QA chain with hybrid retriever created for entire vault");
|
||||
console.log("Set chain:", ChainType.VAULT_QA_CHAIN);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case ChainType.COPILOT_PLUS_CHAIN: {
|
||||
// For initial load of the plugin
|
||||
await this.initializeQAChain(options);
|
||||
this.chain = ChainFactory.createNewLLMChain({
|
||||
llm: chatModel,
|
||||
memory: memory,
|
||||
prompt: options.prompt || chatPrompt,
|
||||
abortController: options.abortController,
|
||||
}) as RunnableSequence;
|
||||
|
||||
setChainType(ChainType.COPILOT_PLUS_CHAIN);
|
||||
break;
|
||||
}
|
||||
|
||||
case ChainType.PROJECT_CHAIN: {
|
||||
// For initial load of the plugin
|
||||
await this.initializeQAChain(options);
|
||||
this.chain = ChainFactory.createNewLLMChain({
|
||||
llm: chatModel,
|
||||
memory: memory,
|
||||
prompt: options.prompt || chatPrompt,
|
||||
abortController: options.abortController,
|
||||
}) as RunnableSequence;
|
||||
setChainType(ChainType.PROJECT_CHAIN);
|
||||
break;
|
||||
}
|
||||
|
||||
default:
|
||||
this.validateChainType(chainType);
|
||||
break;
|
||||
logInfo(`modelKey: ${newModelKey || getModelKey()}`);
|
||||
}
|
||||
}
|
||||
|
||||
private getChainRunner(): ChainRunner {
|
||||
const chainType = getChainType();
|
||||
const settings = getSettings();
|
||||
|
||||
switch (chainType) {
|
||||
case ChainType.LLM_CHAIN:
|
||||
return new LLMChainRunner(this);
|
||||
case ChainType.VAULT_QA_CHAIN:
|
||||
return new VaultQAChainRunner(this);
|
||||
case ChainType.COPILOT_PLUS_CHAIN:
|
||||
// Use AutonomousAgentChainRunner if the setting is enabled
|
||||
if (settings.enableAutonomousAgent) {
|
||||
return new AutonomousAgentChainRunner(this);
|
||||
}
|
||||
return new CopilotPlusChainRunner(this);
|
||||
case ChainType.PROJECT_CHAIN:
|
||||
return new ProjectChainRunner(this);
|
||||
default:
|
||||
throw new Error(`Unsupported chain type: ${chainType}`);
|
||||
throw new Error(`Unsupported chain type: ${String(chainType)}`);
|
||||
}
|
||||
}
|
||||
|
||||
private async initializeQAChain(options: SetChainOptions) {
|
||||
// Handle index refresh if needed
|
||||
if (options.refreshIndex) {
|
||||
await this.vectorStoreManager.indexVaultToVectorStore();
|
||||
}
|
||||
/**
|
||||
* Re-index the vault into the Orama vector store. No-op when legacy
|
||||
* semantic search is disabled — v3 lexical search builds its index on
|
||||
* demand and doesn't need a precomputed store.
|
||||
*/
|
||||
private async refreshVaultIndex() {
|
||||
if (!getSettings().enableSemanticSearchV3) return;
|
||||
const VectorStoreManager = (await import("@/search/vectorStoreManager")).default;
|
||||
await VectorStoreManager.getInstance().indexVaultToVectorStore(false);
|
||||
}
|
||||
|
||||
async runChain(
|
||||
|
|
@ -306,12 +212,15 @@ export default class ChainManager {
|
|||
updateLoading?: (loading: boolean) => void;
|
||||
} = {}
|
||||
) {
|
||||
const { debug = false, ignoreSystemMessage = false } = options;
|
||||
const { ignoreSystemMessage = false } = options;
|
||||
|
||||
if (debug) console.log("==== Step 0: Initial user message ====\n", userMessage);
|
||||
const l5Text = userMessage.contextEnvelope?.layers.find((l) => l.id === "L5_USER")?.text;
|
||||
logInfo(
|
||||
"Step 0: Initial user message:\n",
|
||||
l5Text || userMessage.originalMessage || userMessage.message
|
||||
);
|
||||
|
||||
this.validateChatModel();
|
||||
this.validateChainInitialization();
|
||||
|
||||
const chatModel = this.chatModelManager.getChatModel();
|
||||
|
||||
|
|
@ -331,7 +240,9 @@ export default class ChainManager {
|
|||
]);
|
||||
}
|
||||
|
||||
this.createChainWithNewModel({ prompt: effectivePrompt }, false);
|
||||
void this.createChainWithNewModel({ prompt: effectivePrompt }, false).catch((err) =>
|
||||
logError("createChainWithNewModel failed", err)
|
||||
);
|
||||
/*this.setChain(getChainType(), {
|
||||
prompt: effectivePrompt,
|
||||
});*/
|
||||
|
|
@ -346,33 +257,4 @@ export default class ChainManager {
|
|||
options
|
||||
);
|
||||
}
|
||||
|
||||
async updateMemoryWithLoadedMessages(messages: ChatMessage[]) {
|
||||
await this.memoryManager.clearChatMemory();
|
||||
for (let i = 0; i < messages.length; i += 2) {
|
||||
const userMsg = messages[i];
|
||||
const aiMsg = messages[i + 1];
|
||||
if (userMsg && aiMsg && userMsg.sender === USER_SENDER) {
|
||||
await this.memoryManager
|
||||
.getMemory()
|
||||
.saveContext({ input: userMsg.message }, { output: aiMsg.message });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
public clearHistory() {
|
||||
this.chatMessages = [];
|
||||
}
|
||||
|
||||
public getChatMessages(): ChatMessage[] {
|
||||
return this.chatMessages;
|
||||
}
|
||||
|
||||
public setChatMessages(messages: ChatMessage[]) {
|
||||
this.chatMessages = [...messages];
|
||||
}
|
||||
|
||||
public addChatMessage(message: ChatMessage) {
|
||||
this.chatMessages.push(message);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,983 +0,0 @@
|
|||
import { getCurrentProject } from "@/aiParams";
|
||||
import { getStandaloneQuestion } from "@/chainUtils";
|
||||
import {
|
||||
ABORT_REASON,
|
||||
AI_SENDER,
|
||||
EMPTY_INDEX_ERROR_MESSAGE,
|
||||
LOADING_MESSAGES,
|
||||
MAX_CHARS_FOR_LOCAL_SEARCH_CONTEXT,
|
||||
ModelCapability,
|
||||
} from "@/constants";
|
||||
import {
|
||||
ImageBatchProcessor,
|
||||
ImageContent,
|
||||
ImageProcessingResult,
|
||||
MessageContent,
|
||||
} from "@/imageProcessing/imageProcessor";
|
||||
import { BrevilabsClient } from "@/LLMProviders/brevilabsClient";
|
||||
import { logError, logInfo, logWarn } from "@/logger";
|
||||
import { HybridRetriever } from "@/search/hybridRetriever";
|
||||
import { getSettings, getSystemPrompt } from "@/settings/model";
|
||||
import { ChatMessage } from "@/sharedState";
|
||||
import { ToolManager } from "@/tools/toolManager";
|
||||
import {
|
||||
err2String,
|
||||
extractChatHistory,
|
||||
extractUniqueTitlesFromDocs,
|
||||
extractYoutubeUrl,
|
||||
formatDateTime,
|
||||
getApiErrorMessage,
|
||||
getMessageRole,
|
||||
withSuppressedTokenWarnings,
|
||||
} from "@/utils";
|
||||
import { BaseChatModel } from "@langchain/core/language_models/chat_models";
|
||||
import { Notice } from "obsidian";
|
||||
import ChainManager from "./chainManager";
|
||||
import { COPILOT_TOOL_NAMES, IntentAnalyzer } from "./intentAnalyzer";
|
||||
import ProjectManager from "./projectManager";
|
||||
|
||||
class ThinkBlockStreamer {
|
||||
private hasOpenThinkBlock = false;
|
||||
private fullResponse = "";
|
||||
|
||||
constructor(private updateCurrentAiMessage: (message: string) => void) {}
|
||||
|
||||
private handleClaude37Chunk(content: any[]) {
|
||||
let textContent = "";
|
||||
for (const item of content) {
|
||||
switch (item.type) {
|
||||
case "text":
|
||||
textContent += item.text;
|
||||
break;
|
||||
case "thinking":
|
||||
if (!this.hasOpenThinkBlock) {
|
||||
this.fullResponse += "\n<think>";
|
||||
this.hasOpenThinkBlock = true;
|
||||
}
|
||||
this.fullResponse += item.thinking;
|
||||
this.updateCurrentAiMessage(this.fullResponse);
|
||||
return true; // Indicate we handled a thinking chunk
|
||||
}
|
||||
}
|
||||
if (textContent) {
|
||||
this.fullResponse += textContent;
|
||||
}
|
||||
return false; // No thinking chunk handled
|
||||
}
|
||||
|
||||
private handleDeepseekChunk(chunk: any) {
|
||||
// Handle standard string content
|
||||
if (typeof chunk.content === "string") {
|
||||
this.fullResponse += chunk.content;
|
||||
}
|
||||
|
||||
// Handle deepseek reasoning/thinking content
|
||||
if (chunk.additional_kwargs?.reasoning_content) {
|
||||
if (!this.hasOpenThinkBlock) {
|
||||
this.fullResponse += "\n<think>";
|
||||
this.hasOpenThinkBlock = true;
|
||||
}
|
||||
this.fullResponse += chunk.additional_kwargs.reasoning_content;
|
||||
return true; // Indicate we handled a thinking chunk
|
||||
}
|
||||
return false; // No thinking chunk handled
|
||||
}
|
||||
|
||||
processChunk(chunk: any) {
|
||||
let handledThinking = false;
|
||||
|
||||
// Handle Claude 3.7 array-based content
|
||||
if (Array.isArray(chunk.content)) {
|
||||
handledThinking = this.handleClaude37Chunk(chunk.content);
|
||||
} else {
|
||||
// Handle deepseek format
|
||||
handledThinking = this.handleDeepseekChunk(chunk);
|
||||
}
|
||||
|
||||
// Close think block if we have one open and didn't handle thinking content
|
||||
if (this.hasOpenThinkBlock && !handledThinking) {
|
||||
this.fullResponse += "</think>";
|
||||
this.hasOpenThinkBlock = false;
|
||||
}
|
||||
|
||||
this.updateCurrentAiMessage(this.fullResponse);
|
||||
}
|
||||
|
||||
close() {
|
||||
// Make sure to close any open think block at the end
|
||||
if (this.hasOpenThinkBlock) {
|
||||
this.fullResponse += "</think>";
|
||||
this.updateCurrentAiMessage(this.fullResponse);
|
||||
}
|
||||
return this.fullResponse;
|
||||
}
|
||||
}
|
||||
|
||||
export interface ChainRunner {
|
||||
run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string>;
|
||||
}
|
||||
|
||||
abstract class BaseChainRunner implements ChainRunner {
|
||||
protected chainManager: ChainManager;
|
||||
|
||||
constructor(chainManager: ChainManager) {
|
||||
this.chainManager = chainManager;
|
||||
}
|
||||
|
||||
abstract run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string>;
|
||||
|
||||
protected async handleResponse(
|
||||
fullAIResponse: string,
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
sources?: { title: string; score: number }[]
|
||||
) {
|
||||
// Save to memory and add message if we have a response
|
||||
// Skip only if it's a NEW_CHAT abort (clearing everything)
|
||||
if (
|
||||
fullAIResponse &&
|
||||
!(abortController.signal.aborted && abortController.signal.reason === ABORT_REASON.NEW_CHAT)
|
||||
) {
|
||||
await this.chainManager.memoryManager
|
||||
.getMemory()
|
||||
.saveContext({ input: userMessage.message }, { output: fullAIResponse });
|
||||
|
||||
addMessage({
|
||||
message: fullAIResponse,
|
||||
sender: AI_SENDER,
|
||||
isVisible: true,
|
||||
timestamp: formatDateTime(new Date()),
|
||||
sources: sources,
|
||||
});
|
||||
|
||||
// Clear the streaming message since it's now in chat history
|
||||
updateCurrentAiMessage("");
|
||||
} else if (abortController.signal.reason === ABORT_REASON.NEW_CHAT) {
|
||||
// Also clear if it's a new chat
|
||||
updateCurrentAiMessage("");
|
||||
}
|
||||
logInfo(
|
||||
"==== Chat Memory ====\n",
|
||||
(this.chainManager.memoryManager.getMemory().chatHistory as any).messages.map(
|
||||
(m: any) => m.content
|
||||
)
|
||||
);
|
||||
logInfo("==== Final AI Response ====\n", fullAIResponse);
|
||||
return fullAIResponse;
|
||||
}
|
||||
|
||||
protected async handleError(
|
||||
error: any,
|
||||
addMessage?: (message: ChatMessage) => void,
|
||||
updateCurrentAiMessage?: (message: string) => void
|
||||
) {
|
||||
const msg = err2String(error);
|
||||
logError("Error during LLM invocation:", msg);
|
||||
const errorData = error?.response?.data?.error || msg;
|
||||
const errorCode = errorData?.code || msg;
|
||||
let errorMessage = "";
|
||||
|
||||
// Check for specific error messages
|
||||
if (error?.message?.includes("Invalid license key")) {
|
||||
errorMessage = "Invalid Copilot Plus license key. Please check your license key in settings.";
|
||||
} else if (errorCode === "model_not_found") {
|
||||
errorMessage =
|
||||
"You do not have access to this model or the model does not exist, please check with your API provider.";
|
||||
} else {
|
||||
errorMessage = `${errorCode}`;
|
||||
}
|
||||
|
||||
logError(errorData);
|
||||
|
||||
if (addMessage && updateCurrentAiMessage) {
|
||||
updateCurrentAiMessage("");
|
||||
|
||||
// remove langchain troubleshooting URL from error message
|
||||
const ignoreEndIndex = errorMessage.search("Troubleshooting URL");
|
||||
errorMessage = ignoreEndIndex !== -1 ? errorMessage.slice(0, ignoreEndIndex) : errorMessage;
|
||||
|
||||
// add more user guide for invalid API key
|
||||
if (msg.search(/401|invalid|not valid/gi) !== -1) {
|
||||
errorMessage =
|
||||
"Something went wrong. Please check if you have set your API key." +
|
||||
"\nPath: Settings > copilot plugin > Basic Tab > Set Keys." +
|
||||
"\nOr check model config" +
|
||||
"\nError Details: " +
|
||||
errorMessage;
|
||||
}
|
||||
|
||||
addMessage({
|
||||
message: errorMessage,
|
||||
isErrorMessage: true,
|
||||
sender: AI_SENDER,
|
||||
isVisible: true,
|
||||
timestamp: formatDateTime(new Date()),
|
||||
});
|
||||
} else {
|
||||
// Fallback to Notice if message handlers aren't provided
|
||||
new Notice(errorMessage);
|
||||
logError(errorData);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
class LLMChainRunner extends BaseChainRunner {
|
||||
async run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string> {
|
||||
const streamer = new ThinkBlockStreamer(updateCurrentAiMessage);
|
||||
|
||||
try {
|
||||
// Get chat history from memory
|
||||
const memory = this.chainManager.memoryManager.getMemory();
|
||||
const memoryVariables = await memory.loadMemoryVariables({});
|
||||
const chatHistory = extractChatHistory(memoryVariables);
|
||||
|
||||
// Create messages array starting with system message
|
||||
const messages: any[] = [];
|
||||
|
||||
// Add system message if available
|
||||
const systemPrompt = getSystemPrompt();
|
||||
const chatModel = this.chainManager.chatModelManager.getChatModel();
|
||||
|
||||
if (systemPrompt) {
|
||||
messages.push({
|
||||
role: getMessageRole(chatModel),
|
||||
content: systemPrompt,
|
||||
});
|
||||
}
|
||||
|
||||
// Add chat history
|
||||
for (const entry of chatHistory) {
|
||||
messages.push({ role: entry.role, content: entry.content });
|
||||
}
|
||||
|
||||
// Add current user message
|
||||
messages.push({
|
||||
role: "user",
|
||||
content: userMessage.message,
|
||||
});
|
||||
|
||||
logInfo("==== Final Request to AI ====\n", messages);
|
||||
|
||||
// Stream with abort signal
|
||||
const chatStream = await withSuppressedTokenWarnings(() =>
|
||||
this.chainManager.chatModelManager.getChatModel().stream(messages, {
|
||||
signal: abortController.signal,
|
||||
})
|
||||
);
|
||||
|
||||
for await (const chunk of chatStream) {
|
||||
if (abortController.signal.aborted) {
|
||||
logInfo("Stream iteration aborted", { reason: abortController.signal.reason });
|
||||
break;
|
||||
}
|
||||
streamer.processChunk(chunk);
|
||||
}
|
||||
} catch (error: any) {
|
||||
// Check if the error is due to abort signal
|
||||
if (error.name === "AbortError" || abortController.signal.aborted) {
|
||||
logInfo("Stream aborted by user", { reason: abortController.signal.reason });
|
||||
// Don't show error message for user-initiated aborts
|
||||
} else {
|
||||
await this.handleError(error, addMessage, updateCurrentAiMessage);
|
||||
}
|
||||
}
|
||||
|
||||
// Always return the response, even if partial
|
||||
const response = streamer.close();
|
||||
|
||||
// Only skip saving if it's a new chat (clearing everything)
|
||||
if (abortController.signal.aborted && abortController.signal.reason === ABORT_REASON.NEW_CHAT) {
|
||||
updateCurrentAiMessage("");
|
||||
return "";
|
||||
}
|
||||
|
||||
return this.handleResponse(
|
||||
response,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
class VaultQAChainRunner extends BaseChainRunner {
|
||||
async run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string> {
|
||||
const streamer = new ThinkBlockStreamer(updateCurrentAiMessage);
|
||||
|
||||
try {
|
||||
// Add check for empty index
|
||||
const indexEmpty = await this.chainManager.vectorStoreManager.isIndexEmpty();
|
||||
if (indexEmpty) {
|
||||
return this.handleResponse(
|
||||
EMPTY_INDEX_ERROR_MESSAGE,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage
|
||||
);
|
||||
}
|
||||
|
||||
// Get chat history from memory
|
||||
const memory = this.chainManager.memoryManager.getMemory();
|
||||
const memoryVariables = await memory.loadMemoryVariables({});
|
||||
const chatHistory = extractChatHistory(memoryVariables);
|
||||
|
||||
// Generate standalone question from user message + chat history
|
||||
// This is similar to what the conversational retrieval chain does
|
||||
let standaloneQuestion = userMessage.message;
|
||||
if (chatHistory.length > 0) {
|
||||
// For simplicity, we'll use the original question directly
|
||||
// The original chain would rephrase it, but this approach should work for most cases
|
||||
standaloneQuestion = userMessage.message;
|
||||
}
|
||||
|
||||
// Create retriever (similar to how it's done in chainManager)
|
||||
const retriever = new HybridRetriever({
|
||||
minSimilarityScore: 0.01,
|
||||
maxK: getSettings().maxSourceChunks,
|
||||
salientTerms: [],
|
||||
});
|
||||
|
||||
// Retrieve relevant documents
|
||||
const retrievedDocs = await retriever.getRelevantDocuments(standaloneQuestion);
|
||||
|
||||
// Store retrieved documents for sources
|
||||
this.chainManager.storeRetrieverDocuments(retrievedDocs);
|
||||
|
||||
// Format documents as context
|
||||
const context = retrievedDocs.map((doc: any) => doc.pageContent).join("\n\n");
|
||||
|
||||
// Create messages array
|
||||
const messages: any[] = [];
|
||||
|
||||
// Add system message with QA instruction
|
||||
const systemPrompt = getSystemPrompt();
|
||||
const qaInstructions =
|
||||
"\n\nAnswer the question with as detailed as possible based only on the following context:\n" +
|
||||
context;
|
||||
const fullSystemMessage = systemPrompt + qaInstructions;
|
||||
|
||||
const chatModel = this.chainManager.chatModelManager.getChatModel();
|
||||
if (fullSystemMessage) {
|
||||
messages.push({
|
||||
role: getMessageRole(chatModel),
|
||||
content: fullSystemMessage,
|
||||
});
|
||||
}
|
||||
|
||||
// Add chat history
|
||||
for (const entry of chatHistory) {
|
||||
messages.push({ role: entry.role, content: entry.content });
|
||||
}
|
||||
|
||||
// Add current user question
|
||||
messages.push({
|
||||
role: "user",
|
||||
content: userMessage.message,
|
||||
});
|
||||
|
||||
logInfo("==== Final Request to AI ====\n", messages);
|
||||
|
||||
// Stream with abort signal
|
||||
const chatStream = await withSuppressedTokenWarnings(() =>
|
||||
this.chainManager.chatModelManager.getChatModel().stream(messages, {
|
||||
signal: abortController.signal,
|
||||
})
|
||||
);
|
||||
|
||||
for await (const chunk of chatStream) {
|
||||
if (abortController.signal.aborted) {
|
||||
logInfo("VaultQA stream iteration aborted", { reason: abortController.signal.reason });
|
||||
break;
|
||||
}
|
||||
streamer.processChunk(chunk);
|
||||
}
|
||||
} catch (error: any) {
|
||||
// Check if the error is due to abort signal
|
||||
if (error.name === "AbortError" || abortController.signal.aborted) {
|
||||
logInfo("VaultQA stream aborted by user", { reason: abortController.signal.reason });
|
||||
// Don't show error message for user-initiated aborts
|
||||
} else {
|
||||
await this.handleError(error, addMessage, updateCurrentAiMessage);
|
||||
}
|
||||
}
|
||||
|
||||
// Always get the response, even if partial
|
||||
let fullAIResponse = streamer.close();
|
||||
|
||||
// Only skip saving if it's a new chat (clearing everything)
|
||||
if (abortController.signal.aborted && abortController.signal.reason === ABORT_REASON.NEW_CHAT) {
|
||||
updateCurrentAiMessage("");
|
||||
return "";
|
||||
}
|
||||
|
||||
// Add sources to the response
|
||||
fullAIResponse = this.addSourcestoResponse(fullAIResponse);
|
||||
|
||||
return this.handleResponse(
|
||||
fullAIResponse,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage
|
||||
);
|
||||
}
|
||||
|
||||
private addSourcestoResponse(response: string): string {
|
||||
const docTitles = extractUniqueTitlesFromDocs(this.chainManager.getRetrievedDocuments());
|
||||
if (docTitles.length > 0) {
|
||||
const links = docTitles.map((title) => `- [[${title}]]`).join("\n");
|
||||
response += "\n\n#### Sources:\n\n" + links;
|
||||
}
|
||||
return response;
|
||||
}
|
||||
}
|
||||
|
||||
class CopilotPlusChainRunner extends BaseChainRunner {
|
||||
private isYoutubeOnlyMessage(message: string): boolean {
|
||||
const trimmedMessage = message.trim();
|
||||
const hasYoutubeCommand = trimmedMessage.includes("@youtube");
|
||||
const youtubeUrl = extractYoutubeUrl(trimmedMessage);
|
||||
|
||||
// Check if message only contains @youtube command and a valid URL
|
||||
const words = trimmedMessage
|
||||
.split(/\s+/)
|
||||
.filter((word) => word !== "@youtube" && word.length > 0);
|
||||
|
||||
return hasYoutubeCommand && youtubeUrl !== null && words.length === 1;
|
||||
}
|
||||
|
||||
private async processImageUrls(urls: string[]): Promise<ImageProcessingResult> {
|
||||
const failedImages: string[] = [];
|
||||
const processedImages = await ImageBatchProcessor.processUrlBatch(
|
||||
urls,
|
||||
failedImages,
|
||||
this.chainManager.app.vault
|
||||
);
|
||||
ImageBatchProcessor.showFailedImagesNotice(failedImages);
|
||||
return processedImages;
|
||||
}
|
||||
|
||||
private async processChatInputImages(content: MessageContent[]): Promise<ImageProcessingResult> {
|
||||
const failedImages: string[] = [];
|
||||
const processedImages = await ImageBatchProcessor.processChatImageBatch(
|
||||
content,
|
||||
failedImages,
|
||||
this.chainManager.app.vault
|
||||
);
|
||||
ImageBatchProcessor.showFailedImagesNotice(failedImages);
|
||||
return processedImages;
|
||||
}
|
||||
|
||||
private async extractEmbeddedImages(content: string): Promise<string[]> {
|
||||
const imageRegex = /!\[\[(.*?\.(png|jpg|jpeg|gif|webp|bmp|svg))\]\]/g;
|
||||
const matches = [...content.matchAll(imageRegex)];
|
||||
const images = matches.map((match) => match[1]);
|
||||
return images;
|
||||
}
|
||||
|
||||
private async buildMessageContent(
|
||||
textContent: string,
|
||||
userMessage: ChatMessage
|
||||
): Promise<MessageContent[]> {
|
||||
const failureMessages: string[] = [];
|
||||
const successfulImages: ImageContent[] = [];
|
||||
const settings = getSettings();
|
||||
|
||||
// Collect all image sources
|
||||
const imageSources: { urls: string[]; type: string }[] = [];
|
||||
|
||||
// Safely check and add context URLs
|
||||
const contextUrls = userMessage.context?.urls;
|
||||
if (contextUrls && contextUrls.length > 0) {
|
||||
imageSources.push({ urls: contextUrls, type: "context" });
|
||||
}
|
||||
|
||||
// Process embedded images only if setting is enabled
|
||||
if (settings.passMarkdownImages) {
|
||||
const embeddedImages = await this.extractEmbeddedImages(textContent);
|
||||
if (embeddedImages.length > 0) {
|
||||
imageSources.push({ urls: embeddedImages, type: "embedded" });
|
||||
}
|
||||
}
|
||||
|
||||
// Process all image sources
|
||||
for (const source of imageSources) {
|
||||
const result = await this.processImageUrls(source.urls);
|
||||
successfulImages.push(...result.successfulImages);
|
||||
failureMessages.push(...result.failureDescriptions);
|
||||
}
|
||||
|
||||
// Process existing chat content images if present
|
||||
const existingContent = userMessage.content;
|
||||
if (existingContent && existingContent.length > 0) {
|
||||
const result = await this.processChatInputImages(existingContent);
|
||||
successfulImages.push(...result.successfulImages);
|
||||
failureMessages.push(...result.failureDescriptions);
|
||||
}
|
||||
|
||||
// Let the LLM know about the image processing failures
|
||||
let finalText = textContent;
|
||||
if (failureMessages.length > 0) {
|
||||
finalText = `${textContent}\n\nNote: \n${failureMessages.join("\n")}\n`;
|
||||
}
|
||||
|
||||
const messageContent: MessageContent[] = [
|
||||
{
|
||||
type: "text",
|
||||
text: finalText,
|
||||
},
|
||||
];
|
||||
|
||||
// Add successful images after the text content
|
||||
if (successfulImages.length > 0) {
|
||||
messageContent.push(...successfulImages);
|
||||
}
|
||||
|
||||
return messageContent;
|
||||
}
|
||||
|
||||
private hasCapability(model: BaseChatModel, capability: ModelCapability): boolean {
|
||||
const modelName = (model as any).modelName || (model as any).model || "";
|
||||
const customModel = this.chainManager.chatModelManager.findModelByName(modelName);
|
||||
return customModel?.capabilities?.includes(capability) ?? false;
|
||||
}
|
||||
|
||||
private isMultimodalModel(model: BaseChatModel): boolean {
|
||||
return this.hasCapability(model, ModelCapability.VISION);
|
||||
}
|
||||
|
||||
private async streamMultimodalResponse(
|
||||
textContent: string,
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void
|
||||
): Promise<string> {
|
||||
// Get chat history
|
||||
const memory = this.chainManager.memoryManager.getMemory();
|
||||
const memoryVariables = await memory.loadMemoryVariables({});
|
||||
const chatHistory = extractChatHistory(memoryVariables);
|
||||
|
||||
// Create messages array starting with system message
|
||||
const messages: any[] = [];
|
||||
|
||||
// Add system message if available
|
||||
let fullSystemMessage = await this.getSystemPrompt();
|
||||
|
||||
// Add chat history context to system message if exists
|
||||
if (chatHistory.length > 0) {
|
||||
fullSystemMessage +=
|
||||
"\n\nThe following is the relevant conversation history. Use this context to maintain consistency in your responses:";
|
||||
}
|
||||
|
||||
// Get chat model for role determination for O-series models
|
||||
const chatModel = this.chainManager.chatModelManager.getChatModel();
|
||||
|
||||
// Add the combined system message with appropriate role
|
||||
if (fullSystemMessage) {
|
||||
messages.push({
|
||||
role: getMessageRole(chatModel),
|
||||
content: `${fullSystemMessage}\nIMPORTANT: Maintain consistency with previous responses in the conversation. If you've provided information about a person or topic before, use that same information in follow-up questions.`,
|
||||
});
|
||||
}
|
||||
|
||||
// Add chat history
|
||||
for (const entry of chatHistory) {
|
||||
messages.push({ role: entry.role, content: entry.content });
|
||||
}
|
||||
|
||||
// Get the current chat model
|
||||
const chatModelCurrent = this.chainManager.chatModelManager.getChatModel();
|
||||
const isMultimodalCurrent = this.isMultimodalModel(chatModelCurrent);
|
||||
|
||||
// Build message content with text and images for multimodal models, or just text for text-only models
|
||||
const content = isMultimodalCurrent
|
||||
? await this.buildMessageContent(textContent, userMessage)
|
||||
: textContent;
|
||||
|
||||
// Add current user message
|
||||
messages.push({
|
||||
role: "user",
|
||||
content,
|
||||
});
|
||||
|
||||
const enhancedUserMessage = content instanceof Array ? (content[0] as any).text : content;
|
||||
logInfo("Enhanced user message: ", enhancedUserMessage);
|
||||
logInfo("==== Final Request to AI ====\n", messages);
|
||||
const streamer = new ThinkBlockStreamer(updateCurrentAiMessage);
|
||||
|
||||
// Wrap the stream call with warning suppression
|
||||
const chatStream = await withSuppressedTokenWarnings(() =>
|
||||
this.chainManager.chatModelManager.getChatModel().stream(messages, {
|
||||
signal: abortController.signal,
|
||||
})
|
||||
);
|
||||
|
||||
for await (const chunk of chatStream) {
|
||||
if (abortController.signal.aborted) {
|
||||
logInfo("CopilotPlus multimodal stream iteration aborted", {
|
||||
reason: abortController.signal.reason,
|
||||
});
|
||||
break;
|
||||
}
|
||||
streamer.processChunk(chunk);
|
||||
}
|
||||
|
||||
return streamer.close();
|
||||
}
|
||||
|
||||
async run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
updateLoadingMessage?: (message: string) => void;
|
||||
}
|
||||
): Promise<string> {
|
||||
const { updateLoadingMessage } = options;
|
||||
let fullAIResponse = "";
|
||||
let sources: { title: string; score: number }[] = [];
|
||||
let currentPartialResponse = "";
|
||||
|
||||
// Wrapper to track partial response
|
||||
const trackAndUpdateAiMessage = (message: string) => {
|
||||
currentPartialResponse = message;
|
||||
updateCurrentAiMessage(message);
|
||||
};
|
||||
|
||||
try {
|
||||
// Check if this is a YouTube-only message
|
||||
if (this.isYoutubeOnlyMessage(userMessage.message)) {
|
||||
const url = extractYoutubeUrl(userMessage.message);
|
||||
const failMessage =
|
||||
"Transcript not available. Only videos with the auto transcript option turned on are supported at the moment.";
|
||||
if (url) {
|
||||
try {
|
||||
const response = await BrevilabsClient.getInstance().youtube4llm(url);
|
||||
if (response.response.transcript) {
|
||||
return this.handleResponse(
|
||||
response.response.transcript,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage
|
||||
);
|
||||
}
|
||||
return this.handleResponse(
|
||||
failMessage,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage
|
||||
);
|
||||
} catch (error) {
|
||||
logError("Error processing YouTube video:", error);
|
||||
return this.handleResponse(
|
||||
failMessage,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
logInfo("==== Step 1: Analyzing intent ====");
|
||||
let toolCalls;
|
||||
// Use the original message for intent analysis
|
||||
const messageForAnalysis = userMessage.originalMessage || userMessage.message;
|
||||
try {
|
||||
toolCalls = await IntentAnalyzer.analyzeIntent(messageForAnalysis);
|
||||
} catch (error: any) {
|
||||
return this.handleResponse(
|
||||
getApiErrorMessage(error),
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage
|
||||
);
|
||||
}
|
||||
|
||||
// Use the same removeAtCommands logic as IntentAnalyzer
|
||||
const cleanedUserMessage = userMessage.message
|
||||
.split(" ")
|
||||
.filter((word) => !COPILOT_TOOL_NAMES.includes(word.toLowerCase()))
|
||||
.join(" ")
|
||||
.trim();
|
||||
|
||||
const toolOutputs = await this.executeToolCalls(toolCalls, updateLoadingMessage);
|
||||
const localSearchResult = toolOutputs.find(
|
||||
(output) => output.tool === "localSearch" && output.output && output.output.length > 0
|
||||
);
|
||||
|
||||
// Format chat history from memory
|
||||
const memory = this.chainManager.memoryManager.getMemory();
|
||||
const memoryVariables = await memory.loadMemoryVariables({});
|
||||
const chatHistory = extractChatHistory(memoryVariables);
|
||||
|
||||
if (localSearchResult) {
|
||||
logInfo("==== Step 2: Processing local search results ====");
|
||||
const documents = JSON.parse(localSearchResult.output);
|
||||
|
||||
logInfo("==== Step 3: Condensing Question ====");
|
||||
const standaloneQuestion = await getStandaloneQuestion(cleanedUserMessage, chatHistory);
|
||||
logInfo("Condensed standalone question: ", standaloneQuestion);
|
||||
|
||||
logInfo("==== Step 4: Preparing context ====");
|
||||
const timeExpression = this.getTimeExpression(toolCalls);
|
||||
const context = this.prepareLocalSearchResult(documents, timeExpression);
|
||||
|
||||
const currentTimeOutputs = toolOutputs.filter((output) => output.tool === "getCurrentTime");
|
||||
const enhancedQuestion = this.prepareEnhancedUserMessage(
|
||||
standaloneQuestion,
|
||||
currentTimeOutputs
|
||||
);
|
||||
|
||||
logInfo(context);
|
||||
logInfo("==== Step 5: Invoking QA Chain ====");
|
||||
const qaPrompt = await this.chainManager.promptManager.getQAPrompt({
|
||||
question: enhancedQuestion,
|
||||
context,
|
||||
systemMessage: "", // System prompt is added separately in streamMultimodalResponse
|
||||
});
|
||||
|
||||
fullAIResponse = await this.streamMultimodalResponse(
|
||||
qaPrompt,
|
||||
userMessage,
|
||||
abortController,
|
||||
trackAndUpdateAiMessage
|
||||
);
|
||||
|
||||
// Append sources to the response
|
||||
sources = this.getSources(documents);
|
||||
} else {
|
||||
// Enhance with tool outputs.
|
||||
const enhancedUserMessage = this.prepareEnhancedUserMessage(
|
||||
cleanedUserMessage,
|
||||
toolOutputs
|
||||
);
|
||||
// If no results, default to LLM Chain
|
||||
logInfo("No local search results. Using standard LLM Chain.");
|
||||
|
||||
fullAIResponse = await this.streamMultimodalResponse(
|
||||
enhancedUserMessage,
|
||||
userMessage,
|
||||
abortController,
|
||||
trackAndUpdateAiMessage
|
||||
);
|
||||
}
|
||||
} catch (error: any) {
|
||||
// Reset loading message to default
|
||||
updateLoadingMessage?.(LOADING_MESSAGES.DEFAULT);
|
||||
|
||||
// Check if the error is due to abort signal
|
||||
if (error.name === "AbortError" || abortController.signal.aborted) {
|
||||
logInfo("CopilotPlus stream aborted by user", { reason: abortController.signal.reason });
|
||||
// Don't show error message for user-initiated aborts
|
||||
} else {
|
||||
await this.handleError(error, addMessage, updateCurrentAiMessage);
|
||||
}
|
||||
}
|
||||
|
||||
// Only skip saving if it's a new chat (clearing everything)
|
||||
if (abortController.signal.aborted && abortController.signal.reason === ABORT_REASON.NEW_CHAT) {
|
||||
updateCurrentAiMessage("");
|
||||
return "";
|
||||
}
|
||||
|
||||
// If aborted but not a new chat, use the partial response
|
||||
if (abortController.signal.aborted && currentPartialResponse) {
|
||||
fullAIResponse = currentPartialResponse;
|
||||
}
|
||||
|
||||
return this.handleResponse(
|
||||
fullAIResponse,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage,
|
||||
sources
|
||||
);
|
||||
}
|
||||
|
||||
private getSources(documents: any): { title: string; score: number }[] {
|
||||
if (!documents || !Array.isArray(documents)) {
|
||||
logWarn("No valid documents provided to getSources");
|
||||
return [];
|
||||
}
|
||||
return this.sortUniqueDocsByScore(documents);
|
||||
}
|
||||
|
||||
private sortUniqueDocsByScore(documents: any[]): any[] {
|
||||
const uniqueDocs = new Map<string, any>();
|
||||
|
||||
// Iterate through all documents
|
||||
for (const doc of documents) {
|
||||
if (!doc.title || (!doc?.score && !doc?.rerank_score)) {
|
||||
logWarn("Invalid document structure:", doc);
|
||||
continue;
|
||||
}
|
||||
|
||||
const currentDoc = uniqueDocs.get(doc.title);
|
||||
const isReranked = doc && "rerank_score" in doc;
|
||||
const docScore = isReranked ? doc.rerank_score : doc.score;
|
||||
|
||||
// If the title doesn't exist in the map, or if the new doc has a higher score, update the map
|
||||
if (!currentDoc || docScore > (currentDoc.score ?? 0)) {
|
||||
uniqueDocs.set(doc.title, {
|
||||
title: doc.title,
|
||||
score: docScore,
|
||||
isReranked: isReranked,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Convert the map values back to an array and sort by score in descending order
|
||||
return Array.from(uniqueDocs.values()).sort((a, b) => (b.score ?? 0) - (a.score ?? 0));
|
||||
}
|
||||
|
||||
private async executeToolCalls(
|
||||
toolCalls: any[],
|
||||
updateLoadingMessage?: (message: string) => void
|
||||
) {
|
||||
const toolOutputs = [];
|
||||
for (const toolCall of toolCalls) {
|
||||
logInfo(`==== Step 2: Calling tool: ${toolCall.tool.name} ====`);
|
||||
if (toolCall.tool.name === "localSearch") {
|
||||
updateLoadingMessage?.(LOADING_MESSAGES.READING_FILES);
|
||||
} else if (toolCall.tool.name === "webSearch") {
|
||||
updateLoadingMessage?.(LOADING_MESSAGES.SEARCHING_WEB);
|
||||
} else if (toolCall.tool.name === "getFileTree") {
|
||||
updateLoadingMessage?.(LOADING_MESSAGES.READING_FILE_TREE);
|
||||
}
|
||||
const output = await ToolManager.callTool(toolCall.tool, toolCall.args);
|
||||
toolOutputs.push({ tool: toolCall.tool.name, output });
|
||||
}
|
||||
return toolOutputs;
|
||||
}
|
||||
|
||||
private prepareEnhancedUserMessage(userMessage: string, toolOutputs: any[]) {
|
||||
let context = "";
|
||||
if (toolOutputs.length > 0) {
|
||||
const validOutputs = toolOutputs.filter((output) => output.output != null);
|
||||
if (validOutputs.length > 0) {
|
||||
context =
|
||||
"\n\n# Additional context:\n\n" +
|
||||
validOutputs
|
||||
.map(
|
||||
(output) =>
|
||||
`<${output.tool}>\n${typeof output.output !== "string" ? JSON.stringify(output.output) : output.output}\n</${output.tool}>`
|
||||
)
|
||||
.join("\n\n");
|
||||
}
|
||||
}
|
||||
return `${userMessage}${context}`;
|
||||
}
|
||||
|
||||
private getTimeExpression(toolCalls: any[]): string {
|
||||
const timeRangeCall = toolCalls.find((call) => call.tool.name === "getTimeRangeMs");
|
||||
return timeRangeCall ? timeRangeCall.args.timeExpression : "";
|
||||
}
|
||||
|
||||
private prepareLocalSearchResult(documents: any[], timeExpression: string): string {
|
||||
// First filter documents with includeInContext
|
||||
const includedDocs = documents.filter((doc) => doc.includeInContext);
|
||||
|
||||
// Calculate total content length
|
||||
const totalLength = includedDocs.reduce((sum, doc) => sum + doc.content.length, 0);
|
||||
|
||||
// If total length exceeds threshold, calculate truncation ratio
|
||||
let truncatedDocs = includedDocs;
|
||||
if (totalLength > MAX_CHARS_FOR_LOCAL_SEARCH_CONTEXT) {
|
||||
const truncationRatio = MAX_CHARS_FOR_LOCAL_SEARCH_CONTEXT / totalLength;
|
||||
logInfo("Truncating documents to fit context length. Truncation ratio:", truncationRatio);
|
||||
truncatedDocs = includedDocs.map((doc) => ({
|
||||
...doc,
|
||||
content: doc.content.slice(0, Math.floor(doc.content.length * truncationRatio)),
|
||||
}));
|
||||
}
|
||||
|
||||
const formattedDocs = truncatedDocs
|
||||
.map((doc: any) => `Note in Vault: ${doc.content}`)
|
||||
.join("\n\n");
|
||||
|
||||
return timeExpression
|
||||
? `Local Search Result for ${timeExpression}:\n${formattedDocs}`
|
||||
: `Local Search Result:\n${formattedDocs}`;
|
||||
}
|
||||
|
||||
protected async getSystemPrompt(): Promise<string> {
|
||||
return getSystemPrompt();
|
||||
}
|
||||
}
|
||||
|
||||
class ProjectChainRunner extends CopilotPlusChainRunner {
|
||||
protected async getSystemPrompt(): Promise<string> {
|
||||
let finalPrompt = getSystemPrompt();
|
||||
const projectConfig = getCurrentProject();
|
||||
if (!projectConfig) {
|
||||
return finalPrompt;
|
||||
}
|
||||
|
||||
// Get context asynchronously
|
||||
const context = await ProjectManager.instance.getProjectContext(projectConfig.id);
|
||||
finalPrompt = `${finalPrompt}\n\n<project_system_prompt>\n${projectConfig.systemPrompt}\n</project_system_prompt>`;
|
||||
|
||||
// TODO: Move project context out of the system prompt and into the user prompt.
|
||||
if (context) {
|
||||
finalPrompt = `${finalPrompt}\n\n <project_context>\n${context}\n</project_context>`;
|
||||
}
|
||||
|
||||
return finalPrompt;
|
||||
}
|
||||
}
|
||||
|
||||
export { CopilotPlusChainRunner, LLMChainRunner, ProjectChainRunner, VaultQAChainRunner };
|
||||
262
src/LLMProviders/chainRunner/AutonomousAgentChainRunner.test.ts
Normal file
|
|
@ -0,0 +1,262 @@
|
|||
import {
|
||||
buildToolCallsFromChunks,
|
||||
accumulateToolCallChunk,
|
||||
ToolCallChunk,
|
||||
} from "./utils/nativeToolCalling";
|
||||
|
||||
/**
|
||||
* Test suite for Gemini tool call name extraction fix (Issue #2233)
|
||||
*
|
||||
* Root cause: Gemini's @langchain/google-genai nests tool call names inside
|
||||
* `functionCall.name` instead of at the top level `name` property. Without
|
||||
* the fallback, all Gemini tool call names are empty, causing
|
||||
* buildToolCallsFromChunks to skip them → treated as "no tool calls" →
|
||||
* empty response since thinking tokens were filtered.
|
||||
*/
|
||||
describe("accumulateToolCallChunk", () => {
|
||||
describe("OpenAI-format chunks (top-level name)", () => {
|
||||
it("should accumulate name from top-level tc.name", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
id: "call_123",
|
||||
name: "localSearch",
|
||||
args: '{"query":',
|
||||
});
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
args: '"test"}',
|
||||
});
|
||||
|
||||
const result = chunks.get(0)!;
|
||||
expect(result.name).toBe("localSearch");
|
||||
expect(result.id).toBe("call_123");
|
||||
expect(result.args).toBe('{"query":"test"}');
|
||||
});
|
||||
|
||||
it("should handle multiple concurrent tool calls", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
accumulateToolCallChunk(chunks, { index: 0, name: "localSearch", args: '{"q":"a"}' });
|
||||
accumulateToolCallChunk(chunks, { index: 1, name: "readNote", args: '{"path":"b"}' });
|
||||
|
||||
expect(chunks.get(0)!.name).toBe("localSearch");
|
||||
expect(chunks.get(1)!.name).toBe("readNote");
|
||||
});
|
||||
});
|
||||
|
||||
describe("Gemini-format chunks (name in functionCall)", () => {
|
||||
it("should extract name from functionCall.name when top-level name is missing", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
// Gemini sends chunks with functionCall.name instead of top-level name
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
id: "call_456",
|
||||
functionCall: { name: "localSearch" },
|
||||
args: '{"query":"test"}',
|
||||
});
|
||||
|
||||
const result = chunks.get(0)!;
|
||||
expect(result.name).toBe("localSearch");
|
||||
expect(result.id).toBe("call_456");
|
||||
expect(result.args).toBe('{"query":"test"}');
|
||||
});
|
||||
|
||||
it("should handle multiple Gemini tool calls", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
functionCall: { name: "localSearch" },
|
||||
args: '{"query":"piano"}',
|
||||
});
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 1,
|
||||
functionCall: { name: "readNote" },
|
||||
args: '{"path":"notes/music.md"}',
|
||||
});
|
||||
|
||||
expect(chunks.get(0)!.name).toBe("localSearch");
|
||||
expect(chunks.get(1)!.name).toBe("readNote");
|
||||
});
|
||||
|
||||
it("should prefer top-level name over functionCall.name", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
name: "topLevel",
|
||||
functionCall: { name: "nested" },
|
||||
args: "{}",
|
||||
});
|
||||
|
||||
// Top-level name takes priority via nullish coalescing (??)
|
||||
expect(chunks.get(0)!.name).toBe("topLevel");
|
||||
});
|
||||
});
|
||||
|
||||
describe("Edge cases", () => {
|
||||
it("should default index to 0 when not provided", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
accumulateToolCallChunk(chunks, { name: "localSearch", args: "{}" });
|
||||
|
||||
expect(chunks.has(0)).toBe(true);
|
||||
expect(chunks.get(0)!.name).toBe("localSearch");
|
||||
});
|
||||
|
||||
it("should handle chunk with no name at all", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
accumulateToolCallChunk(chunks, { index: 0, args: '{"query":"test"}' });
|
||||
|
||||
expect(chunks.get(0)!.name).toBe("");
|
||||
expect(chunks.get(0)!.args).toBe('{"query":"test"}');
|
||||
});
|
||||
|
||||
it("should accumulate args across multiple chunks", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
accumulateToolCallChunk(chunks, { index: 0, name: "localSearch", args: '{"qu' });
|
||||
accumulateToolCallChunk(chunks, { index: 0, args: 'ery":' });
|
||||
accumulateToolCallChunk(chunks, { index: 0, args: '"test"}' });
|
||||
|
||||
expect(chunks.get(0)!.args).toBe('{"query":"test"}');
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildToolCallsFromChunks", () => {
|
||||
it("should build tool calls from properly accumulated chunks", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
chunks.set(0, { id: "call_1", name: "localSearch", args: '{"query":"test"}' });
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].name).toBe("localSearch");
|
||||
expect(result[0].args).toEqual({ query: "test" });
|
||||
expect(result[0].id).toBe("call_1");
|
||||
});
|
||||
|
||||
it("should skip chunks with no name (the bug this fix addresses)", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
// This is what happened before the fix: Gemini chunks had no name
|
||||
// because the accumulator didn't check functionCall.name
|
||||
chunks.set(0, { name: "", args: '{"query":"test"}' });
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
// Empty name → skipped → no tool calls → treated as final response
|
||||
expect(result).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("should handle multiple tool calls", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
chunks.set(0, { id: "call_1", name: "localSearch", args: '{"query":"piano"}' });
|
||||
chunks.set(1, { id: "call_2", name: "readNote", args: '{"path":"notes/music.md"}' });
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
expect(result).toHaveLength(2);
|
||||
expect(result[0].name).toBe("localSearch");
|
||||
expect(result[1].name).toBe("readNote");
|
||||
});
|
||||
|
||||
it("should generate an ID when chunk has no id", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
chunks.set(0, { name: "localSearch", args: '{"query":"test"}' });
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].id).toMatch(/^call_/);
|
||||
});
|
||||
|
||||
it("should handle malformed JSON args gracefully", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
chunks.set(0, { name: "localSearch", args: "not valid json" });
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].name).toBe("localSearch");
|
||||
expect(result[0].args).toEqual({});
|
||||
});
|
||||
|
||||
it("should handle empty args", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
chunks.set(0, { name: "localSearch", args: "" });
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].args).toEqual({});
|
||||
});
|
||||
});
|
||||
|
||||
describe("End-to-end: Gemini streaming → buildToolCallsFromChunks", () => {
|
||||
it("should correctly process Gemini-format chunks through the full pipeline", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
// Simulate Gemini streaming: name comes via functionCall, not top-level
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
id: "call_gemini_1",
|
||||
functionCall: { name: "localSearch" },
|
||||
args: '{"query":"piano notes"}',
|
||||
});
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].name).toBe("localSearch");
|
||||
expect(result[0].args).toEqual({ query: "piano notes" });
|
||||
});
|
||||
|
||||
it("should correctly process OpenAI-format chunks through the full pipeline", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
// Simulate OpenAI streaming: name at top level
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
id: "call_openai_1",
|
||||
name: "localSearch",
|
||||
args: '{"query":"piano notes"}',
|
||||
});
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
expect(result).toHaveLength(1);
|
||||
expect(result[0].name).toBe("localSearch");
|
||||
expect(result[0].args).toEqual({ query: "piano notes" });
|
||||
});
|
||||
|
||||
it("should handle sequential Gemini tool calls (the failing scenario)", () => {
|
||||
const chunks = new Map<number, ToolCallChunk>();
|
||||
|
||||
// This is the exact scenario that was failing:
|
||||
// Gemini 3.1 Pro returns 2 sequential tool calls, but names were dropped
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 0,
|
||||
id: "call_g1",
|
||||
functionCall: { name: "localSearch" },
|
||||
args: '{"query":"search term"}',
|
||||
});
|
||||
accumulateToolCallChunk(chunks, {
|
||||
index: 1,
|
||||
id: "call_g2",
|
||||
functionCall: { name: "readNote" },
|
||||
args: '{"path":"some/note.md"}',
|
||||
});
|
||||
|
||||
const result = buildToolCallsFromChunks(chunks);
|
||||
|
||||
// Both tool calls should be preserved — before the fix, both were dropped
|
||||
expect(result).toHaveLength(2);
|
||||
expect(result[0].name).toBe("localSearch");
|
||||
expect(result[1].name).toBe("readNote");
|
||||
});
|
||||
});
|
||||
1110
src/LLMProviders/chainRunner/AutonomousAgentChainRunner.ts
Normal file
244
src/LLMProviders/chainRunner/BaseChainRunner.ts
Normal file
|
|
@ -0,0 +1,244 @@
|
|||
import { ABORT_REASON, AI_SENDER } from "@/constants";
|
||||
import { logError, logInfo } from "@/logger";
|
||||
import { ChatMessage, ResponseMetadata } from "@/types/message";
|
||||
import { err2String, formatDateTime } from "@/utils";
|
||||
import ChainManager from "../chainManager";
|
||||
|
||||
export interface ChainRunner {
|
||||
run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string>;
|
||||
}
|
||||
|
||||
export abstract class BaseChainRunner implements ChainRunner {
|
||||
protected chainManager: ChainManager;
|
||||
|
||||
constructor(chainManager: ChainManager) {
|
||||
this.chainManager = chainManager;
|
||||
}
|
||||
|
||||
abstract run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string>;
|
||||
|
||||
/**
|
||||
* Handles a completed LLM response by saving conversation memory, updating the chat history, and logging summary details.
|
||||
*
|
||||
* @param fullAIResponse - The final response content to present.
|
||||
* @param userMessage - The originating user message.
|
||||
* @param abortController - Abort controller used to track cancellation reasons.
|
||||
* @param addMessage - Callback to append a message to the UI state.
|
||||
* @param updateCurrentAiMessage - Callback to update the streaming message placeholder.
|
||||
* @param sources - Optional sources associated with the response.
|
||||
* @param llmFormattedOutput - Optional formatted output string for memory storage.
|
||||
* @param responseMetadata - Optional metadata describing truncation or token usage.
|
||||
* @returns The full AI response text.
|
||||
*/
|
||||
protected async handleResponse(
|
||||
fullAIResponse: string,
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
sources?: { title: string; path: string; score: number }[],
|
||||
llmFormattedOutput?: string,
|
||||
responseMetadata?: ResponseMetadata
|
||||
) {
|
||||
// Save to memory and add message if we have a response
|
||||
// Skip only if it's a NEW_CHAT abort (clearing everything)
|
||||
|
||||
// Add message if we have a response OR if response was truncated (even if empty)
|
||||
// This ensures truncation warnings are shown even for empty truncated responses
|
||||
const shouldAddMessage =
|
||||
(fullAIResponse || responseMetadata?.wasTruncated) &&
|
||||
!(abortController.signal.aborted && abortController.signal.reason === ABORT_REASON.NEW_CHAT);
|
||||
|
||||
if (shouldAddMessage) {
|
||||
// Save the expanded user message (L5) to memory — NOT the full processedText.
|
||||
// processedText includes context artifact XML (L3), which already lives in the
|
||||
// envelope's L2/L3 layers. Baking it into L4 chat history would cause
|
||||
// triple-inclusion and waste tokens.
|
||||
// L5 preserves prompt-expanded content (e.g. {include_note_content} placeholders)
|
||||
// while excluding context artifact blocks.
|
||||
const l5Text = userMessage.contextEnvelope?.layers.find((l) => l.id === "L5_USER")?.text;
|
||||
const inputForMemory = l5Text || userMessage.originalMessage || userMessage.message;
|
||||
const outputForMemory =
|
||||
llmFormattedOutput || fullAIResponse || "[Response truncated - no content generated]";
|
||||
await this.chainManager.memoryManager.saveContext(
|
||||
{ input: inputForMemory },
|
||||
{ output: outputForMemory }
|
||||
);
|
||||
|
||||
// For empty truncated responses, show a helpful message
|
||||
const displayMessage =
|
||||
fullAIResponse ||
|
||||
(responseMetadata?.wasTruncated
|
||||
? "_[The response was truncated before any content could be generated. Try increasing the max tokens limit.]_"
|
||||
: "");
|
||||
|
||||
const messageToAdd = {
|
||||
message: displayMessage,
|
||||
sender: AI_SENDER,
|
||||
isVisible: true,
|
||||
timestamp: formatDateTime(new Date()),
|
||||
sources: sources,
|
||||
responseMetadata: responseMetadata,
|
||||
};
|
||||
|
||||
addMessage(messageToAdd);
|
||||
|
||||
// Clear the streaming message since it's now in chat history
|
||||
updateCurrentAiMessage("");
|
||||
} else if (abortController.signal.reason === ABORT_REASON.NEW_CHAT) {
|
||||
// Also clear if it's a new chat
|
||||
updateCurrentAiMessage("");
|
||||
}
|
||||
// Log compact memory summary and a truncated final response (~300 chars)
|
||||
const historyMessages = (
|
||||
this.chainManager.memoryManager.getMemory().chatHistory as { messages?: unknown[] }
|
||||
).messages;
|
||||
logInfo("Chat memory updated:\n", {
|
||||
turns: Array.isArray(historyMessages) ? historyMessages.length : 0,
|
||||
});
|
||||
|
||||
const MAX_LOG_LENGTH = 2000;
|
||||
try {
|
||||
const { parseToolCallMarkers } = await import("./utils/toolCallParser");
|
||||
const parsed = parseToolCallMarkers(fullAIResponse);
|
||||
let textOnly = (parsed.segments as { type: string; content: string }[])
|
||||
.map((seg) => (seg.type === "text" ? seg.content : ""))
|
||||
.join("")
|
||||
.trim();
|
||||
if (!textOnly) textOnly = fullAIResponse || "";
|
||||
const snippet =
|
||||
textOnly.length > MAX_LOG_LENGTH
|
||||
? textOnly.slice(0, MAX_LOG_LENGTH) + "... (truncated)"
|
||||
: textOnly;
|
||||
logInfo("Final AI response (truncated):\n", snippet);
|
||||
} catch {
|
||||
// Fallback: truncate raw response without parsing
|
||||
const s = typeof fullAIResponse === "string" ? fullAIResponse : String(fullAIResponse ?? "");
|
||||
const clipped =
|
||||
s.length > MAX_LOG_LENGTH ? s.slice(0, MAX_LOG_LENGTH) + "... (truncated)" : s;
|
||||
logInfo("Final AI response (truncated):\n", clipped);
|
||||
}
|
||||
return fullAIResponse;
|
||||
}
|
||||
|
||||
/**
|
||||
* Logs provider errors and streams a user-friendly message to the UI.
|
||||
*
|
||||
* @param error - Raw provider error object.
|
||||
* @param processErrorChunk - Callback used to stream error text to the UI.
|
||||
*/
|
||||
protected async handleError(error: unknown, processErrorChunk: (message: string) => void) {
|
||||
const msg = err2String(error);
|
||||
logError("Error during LLM invocation:", msg);
|
||||
const errorData =
|
||||
(error as { response?: { data?: { error?: unknown } } })?.response?.data?.error || msg;
|
||||
const errorCode = (errorData as { code?: string })?.code || msg;
|
||||
let errorMessage = "";
|
||||
|
||||
// Check for specific error messages
|
||||
if ((error as { message?: string })?.message?.includes("Invalid license key")) {
|
||||
errorMessage = "Invalid Copilot Plus license key. Please check your license key in settings.";
|
||||
} else if (errorCode === "model_not_found") {
|
||||
errorMessage =
|
||||
"You do not have access to this model or the model does not exist, please check with your API provider.";
|
||||
} else {
|
||||
errorMessage = `${errorCode}`;
|
||||
}
|
||||
|
||||
logError(errorData);
|
||||
processErrorChunk(this.enhancedErrorMsg(errorMessage, msg, error));
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds an enhanced user-facing error message that includes targeted guidance when authentication failures are detected.
|
||||
*
|
||||
* @param errorMessage - The provider-specific error message or code.
|
||||
* @param msg - Normalized error string used for diagnostics.
|
||||
* @param error - Raw error object returned from the provider SDK.
|
||||
* @returns A formatted error string suitable for streaming to the UI.
|
||||
*/
|
||||
private enhancedErrorMsg(errorMessage: string, msg: string, error: unknown) {
|
||||
// remove langchain troubleshooting URL from error message
|
||||
const ignoreEndIndex = errorMessage.search("Troubleshooting URL");
|
||||
errorMessage = ignoreEndIndex !== -1 ? errorMessage.slice(0, ignoreEndIndex) : errorMessage;
|
||||
|
||||
// add more user guide for invalid API key
|
||||
if (this.isAuthenticationError(error, msg)) {
|
||||
errorMessage =
|
||||
"Something went wrong. Please check if you have set your API key." +
|
||||
"\nPath: Settings > copilot plugin > Basic Tab > Set Keys." +
|
||||
"\nOr check model config" +
|
||||
"\nError Details: " +
|
||||
errorMessage;
|
||||
}
|
||||
return errorMessage;
|
||||
}
|
||||
|
||||
/**
|
||||
* Determines whether an error is likely related to authentication or missing API credentials.
|
||||
*
|
||||
* @param error - Raw provider error object.
|
||||
* @param normalizedMessage - Fallback message string for heuristic matching.
|
||||
* @returns True if the error indicates an authentication problem.
|
||||
*/
|
||||
private isAuthenticationError(error: unknown, normalizedMessage: string): boolean {
|
||||
const responseError = (
|
||||
error as {
|
||||
response?: {
|
||||
status?: number;
|
||||
data?: {
|
||||
error?: { status?: number | string; code?: string; message?: string; type?: string };
|
||||
};
|
||||
};
|
||||
}
|
||||
)?.response;
|
||||
const errorData = responseError?.data?.error ?? (error as { error?: unknown })?.error;
|
||||
const rawStatus = responseError?.status ?? (errorData as { status?: number | string })?.status;
|
||||
const statusCode = typeof rawStatus === "string" ? Number.parseInt(rawStatus, 10) : rawStatus;
|
||||
const errorObject =
|
||||
typeof errorData === "object" && errorData !== null
|
||||
? (errorData as Record<string, unknown>)
|
||||
: undefined;
|
||||
const loweredMessage = (
|
||||
typeof errorObject?.message === "string" ? errorObject.message : normalizedMessage
|
||||
).toLowerCase();
|
||||
const loweredCode = typeof errorObject?.code === "string" ? errorObject.code.toLowerCase() : "";
|
||||
const loweredType = typeof errorObject?.type === "string" ? errorObject.type.toLowerCase() : "";
|
||||
|
||||
if (statusCode === 401) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const authHints = [
|
||||
"api key",
|
||||
"apikey",
|
||||
"unauthorized",
|
||||
"authentication",
|
||||
"invalid authentication",
|
||||
];
|
||||
return authHints.some(
|
||||
(hint) =>
|
||||
loweredMessage.includes(hint) || loweredCode.includes(hint) || loweredType.includes(hint)
|
||||
);
|
||||
}
|
||||
}
|
||||
1284
src/LLMProviders/chainRunner/CopilotPlusChainRunner.ts
Normal file
177
src/LLMProviders/chainRunner/LLMChainRunner.ts
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
import { ABORT_REASON, ModelCapability } from "@/constants";
|
||||
import { LayerToMessagesConverter } from "@/context/LayerToMessagesConverter";
|
||||
import { logInfo } from "@/logger";
|
||||
import { getSettings } from "@/settings/model";
|
||||
import { ChatMessage } from "@/types/message";
|
||||
import { findCustomModel, withSuppressedTokenWarnings } from "@/utils";
|
||||
import { BaseChainRunner } from "./BaseChainRunner";
|
||||
import { loadAndAddChatHistory } from "./utils/chatHistoryUtils";
|
||||
import { recordPromptPayload } from "./utils/promptPayloadRecorder";
|
||||
import { ThinkBlockStreamer } from "./utils/ThinkBlockStreamer";
|
||||
import { getModelKey } from "@/aiParams";
|
||||
|
||||
export class LLMChainRunner extends BaseChainRunner {
|
||||
/**
|
||||
* Construct messages array using envelope-based context (L1-L5 layers)
|
||||
* Requires context envelope - throws error if unavailable
|
||||
*/
|
||||
private async constructMessages(
|
||||
userMessage: ChatMessage
|
||||
): Promise<{ role: string; content: string | unknown[] }[]> {
|
||||
// Require envelope for LLM chain
|
||||
if (!userMessage.contextEnvelope) {
|
||||
throw new Error(
|
||||
"[LLMChainRunner] Context envelope is required but not available. Cannot proceed with LLM chain."
|
||||
);
|
||||
}
|
||||
|
||||
logInfo("[LLMChainRunner] Using envelope-based context");
|
||||
|
||||
// Convert envelope to messages (L1 system + L2+L3+L5 user)
|
||||
const baseMessages = LayerToMessagesConverter.convert(userMessage.contextEnvelope, {
|
||||
includeSystemMessage: true,
|
||||
mergeUserContent: true,
|
||||
debug: false,
|
||||
});
|
||||
|
||||
const messages: { role: string; content: string | unknown[] }[] = [];
|
||||
|
||||
// Add system message (L1)
|
||||
const systemMessage = baseMessages.find((m) => m.role === "system");
|
||||
if (systemMessage) {
|
||||
messages.push(systemMessage);
|
||||
}
|
||||
|
||||
// Add chat history (L4)
|
||||
const memory = this.chainManager.memoryManager.getMemory();
|
||||
await loadAndAddChatHistory(memory, messages);
|
||||
|
||||
// Add user message (L2+L3+L5 merged)
|
||||
const userMessageContent = baseMessages.find((m) => m.role === "user");
|
||||
if (userMessageContent) {
|
||||
// Handle multimodal content if present
|
||||
if (userMessage.content && Array.isArray(userMessage.content)) {
|
||||
// Merge envelope text with multimodal content (images)
|
||||
const updatedContent = userMessage.content.map((item: { type?: string }) => {
|
||||
if (item.type === "text") {
|
||||
return { ...item, text: userMessageContent.content };
|
||||
}
|
||||
return item;
|
||||
});
|
||||
messages.push({
|
||||
role: "user",
|
||||
content: updatedContent,
|
||||
});
|
||||
} else {
|
||||
messages.push(userMessageContent);
|
||||
}
|
||||
}
|
||||
|
||||
return messages;
|
||||
}
|
||||
|
||||
async run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string> {
|
||||
// Check if the current model has reasoning capability
|
||||
const settings = getSettings();
|
||||
const modelKey = getModelKey();
|
||||
let excludeThinking = false;
|
||||
|
||||
try {
|
||||
const currentModel = findCustomModel(modelKey, settings.activeModels);
|
||||
// Exclude thinking blocks if model doesn't have REASONING capability
|
||||
excludeThinking = !currentModel.capabilities?.includes(ModelCapability.REASONING);
|
||||
} catch (error) {
|
||||
// If we can't find the model, default to including thinking blocks
|
||||
logInfo(
|
||||
"Could not determine model capabilities, defaulting to include thinking blocks",
|
||||
error
|
||||
);
|
||||
}
|
||||
|
||||
const streamer = new ThinkBlockStreamer(updateCurrentAiMessage, excludeThinking);
|
||||
|
||||
try {
|
||||
// Construct messages using envelope or legacy approach
|
||||
const messages = await this.constructMessages(userMessage);
|
||||
|
||||
// Record the payload for debugging (includes layered view if envelope available)
|
||||
const chatModel = this.chainManager.chatModelManager.getChatModel();
|
||||
const modelName = (chatModel as { modelName?: string } | undefined)?.modelName;
|
||||
recordPromptPayload({
|
||||
messages,
|
||||
modelName,
|
||||
contextEnvelope: userMessage.contextEnvelope,
|
||||
});
|
||||
|
||||
logInfo("Final Request to AI:\n", messages);
|
||||
|
||||
// Stream with abort signal
|
||||
const chatStream = await withSuppressedTokenWarnings(() =>
|
||||
this.chainManager.chatModelManager.getChatModel().stream(
|
||||
// ProviderMessage[] format matches what getChatModel().stream() accepts at runtime
|
||||
messages as never,
|
||||
{
|
||||
signal: abortController.signal,
|
||||
}
|
||||
)
|
||||
);
|
||||
|
||||
for await (const chunk of chatStream) {
|
||||
if (abortController.signal.aborted) {
|
||||
logInfo("Stream iteration aborted", { reason: abortController.signal.reason });
|
||||
break;
|
||||
}
|
||||
streamer.processChunk(chunk as Parameters<typeof streamer.processChunk>[0]);
|
||||
}
|
||||
} catch (error: unknown) {
|
||||
// Check if the error is due to abort signal
|
||||
const errorName = error instanceof Error ? error.name : "";
|
||||
if (errorName === "AbortError" || abortController.signal.aborted) {
|
||||
logInfo("Stream aborted by user", { reason: abortController.signal.reason });
|
||||
// Don't show error message for user-initiated aborts
|
||||
} else {
|
||||
await this.handleError(
|
||||
error,
|
||||
streamer.processErrorChunk.bind(streamer) as (message: string) => void
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Always return the response, even if partial
|
||||
const result = streamer.close();
|
||||
|
||||
const responseMetadata = {
|
||||
wasTruncated: result.wasTruncated,
|
||||
tokenUsage: result.tokenUsage ?? undefined,
|
||||
};
|
||||
|
||||
// Only skip saving if it's a new chat (clearing everything)
|
||||
if (abortController.signal.aborted && abortController.signal.reason === ABORT_REASON.NEW_CHAT) {
|
||||
updateCurrentAiMessage("");
|
||||
return "";
|
||||
}
|
||||
|
||||
await this.handleResponse(
|
||||
result.content,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage,
|
||||
undefined,
|
||||
undefined,
|
||||
responseMetadata
|
||||
);
|
||||
|
||||
return result.content;
|
||||
}
|
||||
}
|
||||
11
src/LLMProviders/chainRunner/ProjectChainRunner.ts
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
import { CopilotPlusChainRunner } from "./CopilotPlusChainRunner";
|
||||
|
||||
/**
|
||||
* ProjectChainRunner - Chain runner for project-based chats
|
||||
*
|
||||
* Project context is automatically added to L1 via ChatManager.getSystemPromptForMessage()
|
||||
* No override needed - inherits all behavior from CopilotPlusChainRunner
|
||||
*/
|
||||
export class ProjectChainRunner extends CopilotPlusChainRunner {
|
||||
// No overrides needed - project context automatically in L1 via ChatManager
|
||||
}
|
||||
708
src/LLMProviders/chainRunner/README.md
Normal file
|
|
@ -0,0 +1,708 @@
|
|||
# Chain Runner Architecture & Tool Calling System
|
||||
|
||||
This directory contains the refactored chain runner system for Obsidian Copilot, providing multiple chain execution strategies with different tool calling approaches.
|
||||
|
||||
## Overview
|
||||
|
||||
The chain runner system provides two distinct tool calling approaches:
|
||||
|
||||
1. **Copilot Plus** (CopilotPlusChainRunner) - Uses native tool calling for intent analysis
|
||||
2. **Autonomous Agent** (AutonomousAgentChainRunner) - Uses native LangChain tool calling with ReAct pattern
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
chainRunner/
|
||||
├── BaseChainRunner.ts # Abstract base class with shared functionality
|
||||
├── LLMChainRunner.ts # Basic LLM interaction (no tools)
|
||||
├── VaultQAChainRunner.ts # Vault-only Q&A with retrieval
|
||||
├── CopilotPlusChainRunner.ts # Legacy tool calling system
|
||||
├── ProjectChainRunner.ts # Project-aware extension of Plus
|
||||
├── AutonomousAgentChainRunner.ts # Native tool calling with ReAct agent loop
|
||||
├── index.ts # Main exports
|
||||
└── utils/
|
||||
├── ThinkBlockStreamer.ts # Handles thinking content from models
|
||||
├── xmlParsing.ts # XML escape/unescape utilities (for context envelope)
|
||||
├── toolExecution.ts # Tool execution helpers
|
||||
└── modelAdapter.ts # Model-specific adaptations
|
||||
```
|
||||
|
||||
## Tool Calling Systems Comparison
|
||||
|
||||
### 1. Model-Based Tool Planning (CopilotPlusChainRunner)
|
||||
|
||||
**How it works:**
|
||||
|
||||
- Uses chat model with `bindTools()` to plan which tools to call
|
||||
- Model outputs tool calls via native `tool_calls` property on AIMessage
|
||||
- Executes tools synchronously before sending to LLM for final response
|
||||
- Enhances user message with tool outputs as context
|
||||
- Supports `@` commands for explicit tool invocation (`@vault`, `@websearch`, `@memory`)
|
||||
|
||||
**Flow:**
|
||||
|
||||
```
|
||||
User Message → Model Planning → Tool Execution → Enhanced Prompt → LLM Response
|
||||
```
|
||||
|
||||
**Example:**
|
||||
|
||||
```typescript
|
||||
// 1. Plan tools using model
|
||||
const { toolCalls, salientTerms } = await this.planToolCalls(message, chatModel);
|
||||
|
||||
// 2. Process @commands (add localSearch, webSearch, etc. if needed)
|
||||
toolCalls = await this.processAtCommands(message, toolCalls, { salientTerms });
|
||||
|
||||
// 3. Execute tools
|
||||
const toolOutputs = await this.executeToolCalls(toolCalls);
|
||||
|
||||
// 4. Send to LLM
|
||||
const response = await this.streamMultimodalResponse(message, toolOutputs, ...);
|
||||
```
|
||||
|
||||
**Tools Available:**
|
||||
|
||||
- `localSearch` - Search vault content
|
||||
- `webSearch` - Search the web
|
||||
- `getCurrentTime` - Get current time
|
||||
- `getFileTree` - Get file structure
|
||||
- `pomodoroTool` - Pomodoro timer
|
||||
- `youtubeTranscription` - YouTube video transcription
|
||||
|
||||
### 2. Autonomous Agent (AutonomousAgentChainRunner)
|
||||
|
||||
**How it works:**
|
||||
|
||||
- Uses native LangChain tool calling via `bindTools()` with ReAct pattern
|
||||
- AI decides autonomously which tools to use via structured `tool_calls`
|
||||
- Iterative loop where AI can call multiple tools in sequence
|
||||
- Each tool result informs the next decision via `ToolMessage`
|
||||
|
||||
**Flow:**
|
||||
|
||||
```
|
||||
User Message → AI Reasoning → tool_calls → Tool Execution →
|
||||
ToolMessage → AI Analysis → More Tools? → Final Response
|
||||
```
|
||||
|
||||
**Native Tool Call Format:**
|
||||
|
||||
```typescript
|
||||
// AIMessage.tool_calls contains structured tool calls
|
||||
const toolCalls = response.tool_calls; // Array of { name, args, id }
|
||||
|
||||
// Example tool call:
|
||||
{
|
||||
name: "localSearch",
|
||||
args: {
|
||||
query: "machine learning notes",
|
||||
salientTerms: ["machine", "learning", "AI", "algorithms"]
|
||||
},
|
||||
id: "call_abc123"
|
||||
}
|
||||
```
|
||||
|
||||
**ReAct Loop:**
|
||||
|
||||
```typescript
|
||||
// Bind tools to model for native tool calling
|
||||
const boundModel = chatModel.bindTools(availableTools);
|
||||
|
||||
while (iteration < maxIterations) {
|
||||
// 1. Get AI response with potential tool calls
|
||||
const response = await boundModel.invoke(messages);
|
||||
messages.push(response);
|
||||
|
||||
// 2. Check for tool calls in structured format
|
||||
if (!response.tool_calls || response.tool_calls.length === 0) {
|
||||
// No tools needed - final response
|
||||
break;
|
||||
}
|
||||
|
||||
// 3. Execute each tool and add ToolMessage
|
||||
for (const toolCall of response.tool_calls) {
|
||||
const result = await executeSequentialToolCall(toolCall, availableTools);
|
||||
messages.push(
|
||||
new ToolMessage({
|
||||
content: JSON.stringify(result),
|
||||
tool_call_id: toolCall.id,
|
||||
name: toolCall.name,
|
||||
})
|
||||
);
|
||||
}
|
||||
|
||||
// 4. Continue loop - AI sees tool results via ToolMessage
|
||||
}
|
||||
```
|
||||
|
||||
### ReAct Prompting Flow
|
||||
|
||||
Each iteration sends the following message structure to the LLM:
|
||||
|
||||
**Iteration 1 (Initial):**
|
||||
|
||||
```
|
||||
messages = [
|
||||
SystemMessage: "You are a helpful assistant... [tool descriptions via bindTools]"
|
||||
HumanMessage: "What did I write about machine learning last week?"
|
||||
]
|
||||
```
|
||||
|
||||
**Iteration 1 Response:**
|
||||
|
||||
```
|
||||
AIMessage: {
|
||||
content: "", // May be empty or contain reasoning
|
||||
tool_calls: [{
|
||||
id: "call_abc123",
|
||||
name: "getTimeRangeMs",
|
||||
args: { description: "last week" }
|
||||
}]
|
||||
}
|
||||
```
|
||||
|
||||
**Iteration 2 (After Tool Execution):**
|
||||
|
||||
```
|
||||
messages = [
|
||||
SystemMessage: "..."
|
||||
HumanMessage: "What did I write about machine learning last week?"
|
||||
AIMessage: { tool_calls: [getTimeRangeMs] }
|
||||
ToolMessage: { tool_call_id: "call_abc123", content: '{"startTime":1736..., "endTime":1737...}' }
|
||||
]
|
||||
```
|
||||
|
||||
**Iteration 2 Response:**
|
||||
|
||||
```
|
||||
AIMessage: {
|
||||
content: "",
|
||||
tool_calls: [{
|
||||
id: "call_def456",
|
||||
name: "localSearch",
|
||||
args: {
|
||||
query: "machine learning",
|
||||
salientTerms: ["machine learning", "ML", "AI"],
|
||||
timeRange: { startTime: 1736..., endTime: 1737... }
|
||||
}
|
||||
}]
|
||||
}
|
||||
```
|
||||
|
||||
**Iteration 3 (After Second Tool):**
|
||||
|
||||
```
|
||||
messages = [
|
||||
SystemMessage: "..."
|
||||
HumanMessage: "What did I write about machine learning last week?"
|
||||
AIMessage: { tool_calls: [getTimeRangeMs] }
|
||||
ToolMessage: { tool_call_id: "call_abc123", content: '{"startTime":..., "endTime":...}' }
|
||||
AIMessage: { tool_calls: [localSearch] }
|
||||
ToolMessage: { tool_call_id: "call_def456", content: '{"documents": [...5 results...]}' }
|
||||
]
|
||||
```
|
||||
|
||||
**Final Response (No tool_calls):**
|
||||
|
||||
```
|
||||
AIMessage: {
|
||||
content: "Based on your notes from last week, you wrote about...",
|
||||
tool_calls: [] // Empty = final response, exit loop
|
||||
}
|
||||
```
|
||||
|
||||
### Key Points
|
||||
|
||||
1. **Tool schemas** are provided via `bindTools()` - the LLM sees them in its context
|
||||
2. **AIMessage with tool_calls** triggers tool execution; **AIMessage without tool_calls** is final response
|
||||
3. **ToolMessage** correlates with AIMessage via `tool_call_id`
|
||||
4. **Conversation history grows** - each iteration sees all previous messages
|
||||
5. **Max 4 iterations** to prevent infinite loops
|
||||
|
||||
## Key Differences
|
||||
|
||||
| Aspect | Copilot Plus | Autonomous Agent |
|
||||
| ------------------ | ------------------------------- | ------------------------------------- |
|
||||
| **Tool Decision** | Model-based intent planning | AI decides autonomously (ReAct) |
|
||||
| **Tool Execution** | Pre-LLM, synchronous | During conversation, iterative |
|
||||
| **Tool Format** | Native tool calling (bindTools) | Native tool calling (bindTools) |
|
||||
| **Reasoning** | Intent analysis → tools | AI reasoning → tools → more reasoning |
|
||||
| **Iterations** | Single pass | Up to 4 iterations |
|
||||
| **Tool Chaining** | Limited | Full chaining support |
|
||||
|
||||
## LangChain Tool Interface
|
||||
|
||||
### Overview
|
||||
|
||||
Tools are created using native LangChain's `tool()` function via the `createLangChainTool` helper, with Zod schema validation. Tool metadata (execution control, display info) is stored separately in `ToolRegistry`.
|
||||
|
||||
```typescript
|
||||
// Tool creation returns a LangChain StructuredTool
|
||||
const myTool = createLangChainTool({
|
||||
name: string;
|
||||
description: string;
|
||||
schema: z.ZodType;
|
||||
func: (args) => Promise<string | object>;
|
||||
});
|
||||
|
||||
// Tool metadata stored in ToolRegistry
|
||||
interface ToolMetadata {
|
||||
id: string;
|
||||
displayName: string;
|
||||
description: string;
|
||||
category: "search" | "time" | "file" | "media" | "mcp" | "memory" | "custom";
|
||||
isAlwaysEnabled?: boolean;
|
||||
timeoutMs?: number;
|
||||
isBackground?: boolean;
|
||||
isPlusOnly?: boolean;
|
||||
}
|
||||
```
|
||||
|
||||
### Creating Tools
|
||||
|
||||
All tools are created using `createLangChainTool` with Zod schemas:
|
||||
|
||||
#### Tool with No Parameters
|
||||
|
||||
```typescript
|
||||
const indexTool = createLangChainTool({
|
||||
name: "indexVault",
|
||||
description: "Index the vault to the Copilot index",
|
||||
schema: z.object({}), // Empty object for no parameters
|
||||
func: async () => {
|
||||
// Tool implementation
|
||||
return { status: "complete" };
|
||||
},
|
||||
});
|
||||
|
||||
// Register with metadata
|
||||
registry.register({
|
||||
tool: indexTool,
|
||||
metadata: {
|
||||
id: "indexVault",
|
||||
displayName: "Index Vault",
|
||||
description: "Index the vault",
|
||||
category: "file",
|
||||
isBackground: true,
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
#### Tool with Parameters
|
||||
|
||||
```typescript
|
||||
// Define schema with validation rules
|
||||
const searchSchema = z.object({
|
||||
query: z.string().min(1).describe("The search query"),
|
||||
salientTerms: z.array(z.string()).min(1).describe("Key terms extracted from query"),
|
||||
timeRange: z
|
||||
.object({
|
||||
startTime: z.any(),
|
||||
endTime: z.any(),
|
||||
})
|
||||
.optional()
|
||||
.describe("Time range for search"),
|
||||
});
|
||||
|
||||
// Create tool with automatic validation
|
||||
const searchTool = createLangChainTool({
|
||||
name: "localSearch",
|
||||
description: "Search for notes based on query and time range",
|
||||
schema: searchSchema,
|
||||
func: async ({ query, salientTerms, timeRange }) => {
|
||||
// Handler receives fully typed and validated arguments
|
||||
return performSearch(query, salientTerms, timeRange);
|
||||
},
|
||||
});
|
||||
|
||||
// Register with metadata
|
||||
registry.register({
|
||||
tool: searchTool,
|
||||
metadata: {
|
||||
id: "localSearch",
|
||||
displayName: "Vault Search",
|
||||
category: "search",
|
||||
timeoutMs: 30000,
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
### Benefits of LangChain Native Tools
|
||||
|
||||
1. **Type Safety**: Full TypeScript type inference from Zod schemas
|
||||
2. **Runtime Validation**: All inputs validated before reaching handler
|
||||
3. **Native LangChain Integration**: Compatible with `bindTools()` and LangChain tooling ecosystem
|
||||
4. **Better Error Messages**: Zod provides detailed validation errors
|
||||
5. **Separation of Concerns**: Tool implementation separate from execution metadata
|
||||
6. **Future-Proof**: Ready for native tool calling when models support it
|
||||
|
||||
### Advanced Zod Patterns
|
||||
|
||||
#### Complex Validation
|
||||
|
||||
```typescript
|
||||
const emailToolSchema = z.object({
|
||||
to: z.string().email().describe("Recipient email"),
|
||||
subject: z.string().min(1).max(100).describe("Email subject"),
|
||||
body: z.string().min(1).describe("Email content"),
|
||||
cc: z.array(z.string().email()).optional().describe("CC recipients"),
|
||||
});
|
||||
```
|
||||
|
||||
#### Union Types for Actions
|
||||
|
||||
```typescript
|
||||
const actionSchema = z.discriminatedUnion("type", [
|
||||
z.object({
|
||||
type: z.literal("search"),
|
||||
query: z.string().min(1),
|
||||
}),
|
||||
z.object({
|
||||
type: z.literal("create"),
|
||||
content: z.string().min(1),
|
||||
tags: z.array(z.string()).default([]),
|
||||
}),
|
||||
z.object({
|
||||
type: z.literal("delete"),
|
||||
id: z.string().uuid(),
|
||||
}),
|
||||
]);
|
||||
|
||||
const actionTool = createLangChainTool({
|
||||
name: "performAction",
|
||||
description: "Perform various actions",
|
||||
schema: actionSchema,
|
||||
func: async (action) => {
|
||||
// TypeScript knows exactly which type based on discriminator
|
||||
switch (action.type) {
|
||||
case "search":
|
||||
return search(action.query);
|
||||
case "create":
|
||||
return create(action.content, action.tags);
|
||||
case "delete":
|
||||
return deleteItem(action.id);
|
||||
}
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
#### Custom Validation
|
||||
|
||||
```typescript
|
||||
const filePathSchema = z
|
||||
.string()
|
||||
.refine((val) => val.endsWith(".md") || val.endsWith(".canvas"), {
|
||||
message: "File must be .md or .canvas",
|
||||
})
|
||||
.refine((val) => !val.includes(".."), { message: "Path traversal not allowed" })
|
||||
.describe("Path to markdown or canvas file");
|
||||
```
|
||||
|
||||
#### Transformations
|
||||
|
||||
```typescript
|
||||
const dateToolSchema = z.object({
|
||||
date: z
|
||||
.string()
|
||||
.describe("Date in ISO format or natural language")
|
||||
.transform((str) => new Date(str)),
|
||||
timezone: z.string().default("UTC").describe("Timezone identifier"),
|
||||
});
|
||||
```
|
||||
|
||||
### Schema Composition
|
||||
|
||||
```typescript
|
||||
// Base schemas that can be reused
|
||||
const timeRangeSchema = z
|
||||
.object({
|
||||
startTime: z.date(),
|
||||
endTime: z.date(),
|
||||
})
|
||||
.refine((data) => data.endTime > data.startTime, {
|
||||
message: "End time must be after start time",
|
||||
});
|
||||
|
||||
const paginationSchema = z.object({
|
||||
page: z.number().int().positive().default(1),
|
||||
pageSize: z.number().int().positive().max(100).default(20),
|
||||
});
|
||||
|
||||
// Compose into larger schemas
|
||||
const searchWithPaginationSchema = z.object({
|
||||
query: z.string().min(1).describe("Search query"),
|
||||
filters: z.record(z.string()).optional().describe("Additional filters"),
|
||||
timeRange: timeRangeSchema.optional(),
|
||||
pagination: paginationSchema,
|
||||
});
|
||||
```
|
||||
|
||||
### Default Values
|
||||
|
||||
```typescript
|
||||
const configSchema = z.object({
|
||||
temperature: z.number().min(0).max(2).default(0.7),
|
||||
maxTokens: z.number().int().positive().default(1000),
|
||||
model: z.enum(["gpt-4", "gpt-3.5-turbo"]).default("gpt-4"),
|
||||
});
|
||||
|
||||
// Handler receives object with defaults applied
|
||||
const configTool = createLangChainTool({
|
||||
name: "updateConfig",
|
||||
schema: configSchema,
|
||||
func: async (config) => {
|
||||
// config.temperature is always defined (0.7 if not provided)
|
||||
// config.maxTokens is always defined (1000 if not provided)
|
||||
// config.model is always defined ("gpt-4" if not provided)
|
||||
return updateConfiguration(config);
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
### Handling Validation Errors with Retry
|
||||
|
||||
When AI-generated parameters fail Zod validation, the tool execution will return a formatted error. The autonomous agent automatically handles this through its iterative loop:
|
||||
|
||||
```typescript
|
||||
// Example tool with strict validation
|
||||
const searchToolWithValidation = createLangChainTool({
|
||||
name: "searchNotes",
|
||||
description: "Search notes with specific criteria",
|
||||
schema: z.object({
|
||||
query: z.string().min(2, "Query must be at least 2 characters"),
|
||||
limit: z.number().int().min(1).max(100),
|
||||
sortBy: z.enum(["relevance", "date", "title"]),
|
||||
}),
|
||||
func: async ({ query, limit, sortBy }) => {
|
||||
return performSearch(query, limit, sortBy);
|
||||
},
|
||||
});
|
||||
|
||||
// When the AI provides invalid parameters:
|
||||
// Input: { query: "a", limit: 200, sortBy: "random" }
|
||||
//
|
||||
// The flow:
|
||||
// 1. Tool execution catches Zod validation error
|
||||
// 2. Returns: "Tool searchNotes validation failed: query: Query must be at least 2 characters,
|
||||
// limit: Number must be less than or equal to 100, sortBy: Invalid enum value"
|
||||
// 3. This error is added to the conversation as a user message
|
||||
// 4. The AI sees the error in the next iteration and can retry with corrected parameters
|
||||
// 5. The autonomous agent continues up to 4 iterations, allowing multiple retry attempts
|
||||
|
||||
// Example conversation flow:
|
||||
// Iteration 1: AI calls tool with invalid params → receives error
|
||||
// Iteration 2: AI understands error and retries with { query: "search term", limit: 50, sortBy: "date" } → success
|
||||
```
|
||||
|
||||
The validation errors are automatically formatted to be clear and actionable, helping the AI self-correct. The autonomous agent's iterative design naturally provides retry capability with the AI learning from each error.
|
||||
|
||||
## Native Tool Calling Details
|
||||
|
||||
### Tool Execution (`toolExecution.ts`)
|
||||
|
||||
```typescript
|
||||
// Execute individual tool with timeout and error handling
|
||||
async function executeSequentialToolCall(
|
||||
toolCall: ToolCall,
|
||||
availableTools: any[]
|
||||
): Promise<ToolExecutionResult> {
|
||||
// 30-second timeout per tool
|
||||
// Error handling and validation
|
||||
// Result formatting
|
||||
}
|
||||
|
||||
// ToolCall interface (from native tool calling)
|
||||
interface ToolCall {
|
||||
name: string;
|
||||
args: Record<string, unknown>;
|
||||
id?: string; // Used for ToolMessage correlation
|
||||
}
|
||||
```
|
||||
|
||||
### Available Tools in Agent Mode
|
||||
|
||||
All tools from the Copilot Plus system plus autonomous decision-making:
|
||||
|
||||
- **localSearch** - Vault content search with salient terms and query expansion
|
||||
- **webSearch** - Web search with chat history context
|
||||
- **getFileTree** - File structure exploration
|
||||
- **getCurrentTime** / **getTimeRangeMs** - Time-based queries
|
||||
- **pomodoroTool** - Productivity timer
|
||||
- **indexTool** - Vault indexing operations
|
||||
- **youtubeTranscription** - Video content analysis
|
||||
|
||||
### System Prompt Engineering
|
||||
|
||||
The Autonomous Agent mode uses a comprehensive system prompt that:
|
||||
|
||||
1. **Describes available tools** - Tool schemas are provided via `bindTools()`
|
||||
2. **Provides behavioral guidance** - When to use each tool, how to chain them
|
||||
3. **Sets expectations** for reasoning and tool chaining
|
||||
4. **Includes critical requirements** (e.g., salientTerms for localSearch)
|
||||
|
||||
Tool descriptions are provided automatically via Zod schemas. Model adapters add behavioral guidance:
|
||||
|
||||
```typescript
|
||||
// Example: Model adapter adds tool usage guidance
|
||||
enhanceSystemPrompt(basePrompt: string): string {
|
||||
return basePrompt + `
|
||||
|
||||
When searching notes, always provide both "query" (string) and "salientTerms" (array of key terms).
|
||||
Use getTimeRangeMs before localSearch for time-based queries.
|
||||
`;
|
||||
}
|
||||
```
|
||||
|
||||
## Benefits of Autonomous Agent
|
||||
|
||||
1. **Autonomous Tool Selection** - AI decides what tools to use without pre-analysis
|
||||
2. **Tool Chaining** - Can use results from one tool to inform the next
|
||||
3. **Complex Workflows** - Multi-step reasoning with tool support
|
||||
4. **Model Agnostic** - Works with any LLM that supports native tool calling
|
||||
5. **No External Dependencies** - No Brevilabs API required
|
||||
6. **Transparency** - User can see the AI's reasoning process via Agent Reasoning Block
|
||||
7. **Native Integration** - Uses LangChain's `bindTools()` for proper tool calling support
|
||||
|
||||
## Usage
|
||||
|
||||
### Enable Autonomous Agent
|
||||
|
||||
```typescript
|
||||
// In settings
|
||||
settings.enableAutonomousAgent = true;
|
||||
|
||||
// ChainManager automatically selects the appropriate runner
|
||||
const runner = chainManager.getChainRunner(); // Returns AutonomousAgentChainRunner
|
||||
```
|
||||
|
||||
### Example Query Flow
|
||||
|
||||
**User Input:** "Find my notes about machine learning and research current best practices"
|
||||
|
||||
**Autonomous Agent Process:**
|
||||
|
||||
1. **Iteration 1**: AI reasons about the task → calls `localSearch` for ML notes
|
||||
2. **Iteration 2**: Analyzes vault results → calls `webSearch` for current practices
|
||||
3. **Iteration 3**: Synthesizes both sources → provides comprehensive response
|
||||
|
||||
**Legacy Process:**
|
||||
|
||||
1. Intent analysis determines both tools needed
|
||||
2. Executes both tools
|
||||
3. Single LLM call with all context
|
||||
|
||||
## Error Handling & Fallbacks
|
||||
|
||||
### Autonomous Agent Fallbacks
|
||||
|
||||
```typescript
|
||||
try {
|
||||
// Sequential thinking execution
|
||||
} catch (error) {
|
||||
// Automatic fallback to CopilotPlusChainRunner
|
||||
const fallbackRunner = new CopilotPlusChainRunner(this.chainManager);
|
||||
return await fallbackRunner.run(/* same parameters */);
|
||||
}
|
||||
```
|
||||
|
||||
### Tool Execution Safeguards
|
||||
|
||||
- 30-second timeout per tool
|
||||
- Graceful error handling with descriptive messages
|
||||
- Tool availability validation
|
||||
- Result validation and formatting
|
||||
|
||||
## Model Adapter Pattern
|
||||
|
||||
### Overview
|
||||
|
||||
The Model Adapter pattern handles model-specific quirks and requirements cleanly, keeping the core logic model-agnostic.
|
||||
|
||||
### Architecture
|
||||
|
||||
```typescript
|
||||
interface ModelAdapter {
|
||||
enhanceSystemPrompt(basePrompt: string, toolDescriptions: string): string;
|
||||
enhanceUserMessage(message: string, requiresTools: boolean): string;
|
||||
needsSpecialHandling(): boolean;
|
||||
}
|
||||
```
|
||||
|
||||
> **Note:** With native tool calling via `bindTools()`, tool calls are returned in structured `response.tool_calls` format. Model adapters now focus on behavioral guidance rather than parsing or response sanitization.
|
||||
|
||||
### Current Adapters
|
||||
|
||||
1. **BaseModelAdapter** - Default behavior for well-behaved models
|
||||
2. **GPTModelAdapter** - Aggressive prompting for GPT models that often skip tool calls
|
||||
3. **ClaudeModelAdapter** - Specialized handling for Claude thinking models (3.7 Sonnet, Claude 4)
|
||||
4. **GeminiModelAdapter** - Ready for Gemini-specific handling
|
||||
|
||||
### Adding a New Model
|
||||
|
||||
```typescript
|
||||
class NewModelAdapter extends BaseModelAdapter {
|
||||
enhanceSystemPrompt(basePrompt: string, toolDescriptions: string): string {
|
||||
const base = super.enhanceSystemPrompt(basePrompt, toolDescriptions);
|
||||
return base + "\n\n[Model-specific instructions here]";
|
||||
}
|
||||
|
||||
enhanceUserMessage(message: string, requiresTools: boolean): string {
|
||||
// Add model-specific hints if needed
|
||||
return requiresTools ? `${message}\n[Model-specific hint]` : message;
|
||||
}
|
||||
}
|
||||
|
||||
// Register in ModelAdapterFactory
|
||||
if (modelName.includes("newmodel")) {
|
||||
return new NewModelAdapter(modelName);
|
||||
}
|
||||
```
|
||||
|
||||
### Claude Model Adapter Features
|
||||
|
||||
The `ClaudeModelAdapter` includes specialized handling for Claude thinking models:
|
||||
|
||||
#### Thinking Model Support
|
||||
|
||||
- **Claude 3.7 Sonnet** and **Claude 4** - Automatic thinking mode configuration
|
||||
- **Think Block Preservation** - Maintains valuable reasoning context in responses
|
||||
- **Temperature Control** - Disables temperature for thinking models (as required by API)
|
||||
|
||||
> **Note:** With native tool calling, tool calls are returned in structured `response.tool_calls` format. Intermediate tool calls are hidden from users and displayed via the Agent Reasoning Block. Only final responses are streamed to the UI.
|
||||
|
||||
#### Agent Reasoning Block
|
||||
|
||||
The reasoning process is now displayed via the Agent Reasoning Block component, which shows:
|
||||
|
||||
```
|
||||
⏱️ 2.3s elapsed
|
||||
├─ Searching notes for "piano", "learning", "practice"...
|
||||
├─ Found 5 notes: Piano Practice.md, Learning Music.md...
|
||||
└─ Generating response...
|
||||
```
|
||||
|
||||
### Benefits
|
||||
|
||||
1. **Separation of Concerns** - Model quirks isolated from core logic
|
||||
2. **Maintainability** - Easy to find and update model-specific code
|
||||
3. **Extensibility** - Simple to add support for new models
|
||||
4. **Testing** - Model adapters can be unit tested independently
|
||||
5. **Clean Core** - Autonomous agent logic remains model-agnostic
|
||||
6. **Hallucination Prevention** - Specialized handling for problematic models
|
||||
7. **Streaming Protection** - Prevents bad content from reaching users
|
||||
8. **Generalizable Solutions** - Uses threshold-based detection over regex patterns
|
||||
|
||||
## Future Considerations
|
||||
|
||||
1. **Tool Discovery** - Dynamic tool registration
|
||||
2. **Custom Tools** - User-defined tool capabilities
|
||||
3. **Parallel Execution** - Multiple tools simultaneously
|
||||
4. **Tool Result Caching** - Avoid redundant calls
|
||||
5. **Advanced Reasoning** - More sophisticated decision trees
|
||||
6. **Tool Permissions** - User control over tool access (human-in-the-loop approval)
|
||||
7. **Deep Search** - Iterative search refinement for complex queries
|
||||
8. **Response Validation** - Adapters could validate model outputs
|
||||
9. **Model-Specific Optimizations** - Expand adapter capabilities for emerging models
|
||||
|
||||
The autonomous agent approach using native tool calling represents a significant evolution from traditional tool calling, enabling more sophisticated AI reasoning and autonomous task completion.
|
||||
324
src/LLMProviders/chainRunner/VaultQAChainRunner.ts
Normal file
|
|
@ -0,0 +1,324 @@
|
|||
import { ABORT_REASON, ModelCapability, RETRIEVED_DOCUMENT_TAG } from "@/constants";
|
||||
import { getStandaloneQuestion } from "@/chainUtils";
|
||||
import { LayerToMessagesConverter } from "@/context/LayerToMessagesConverter";
|
||||
import { logInfo } from "@/logger";
|
||||
import { RetrieverFactory } from "@/search/RetrieverFactory";
|
||||
import { FilterRetriever } from "@/search/v3/FilterRetriever";
|
||||
import { mergeFilterAndSearchResults } from "@/search/v3/mergeResults";
|
||||
import { extractTagsFromQuery } from "@/search/v3/utils/tagUtils";
|
||||
import { getSettings } from "@/settings/model";
|
||||
import { ChatMessage } from "@/types/message";
|
||||
import {
|
||||
extractChatHistory,
|
||||
extractUniqueTitlesFromDocs,
|
||||
findCustomModel,
|
||||
getMessageRole,
|
||||
withSuppressedTokenWarnings,
|
||||
} from "@/utils";
|
||||
import { BaseChainRunner } from "./BaseChainRunner";
|
||||
import { loadAndAddChatHistory } from "./utils/chatHistoryUtils";
|
||||
import {
|
||||
formatSourceCatalog,
|
||||
getQACitationInstructions,
|
||||
sanitizeContentForCitations,
|
||||
addFallbackSources,
|
||||
hasInlineCitations,
|
||||
type SourceCatalogEntry,
|
||||
} from "./utils/citationUtils";
|
||||
import { recordPromptPayload } from "./utils/promptPayloadRecorder";
|
||||
import { ThinkBlockStreamer } from "./utils/ThinkBlockStreamer";
|
||||
import { getModelKey } from "@/aiParams";
|
||||
|
||||
export class VaultQAChainRunner extends BaseChainRunner {
|
||||
async run(
|
||||
userMessage: ChatMessage,
|
||||
abortController: AbortController,
|
||||
updateCurrentAiMessage: (message: string) => void,
|
||||
addMessage: (message: ChatMessage) => void,
|
||||
options: {
|
||||
debug?: boolean;
|
||||
ignoreSystemMessage?: boolean;
|
||||
updateLoading?: (loading: boolean) => void;
|
||||
}
|
||||
): Promise<string> {
|
||||
// Check if the current model has reasoning capability
|
||||
const settings = getSettings();
|
||||
const modelKey = getModelKey();
|
||||
let excludeThinking = false;
|
||||
|
||||
try {
|
||||
const currentModel = findCustomModel(modelKey, settings.activeModels);
|
||||
// Exclude thinking blocks if model doesn't have REASONING capability
|
||||
excludeThinking = !currentModel.capabilities?.includes(ModelCapability.REASONING);
|
||||
} catch (error) {
|
||||
// If we can't find the model, default to including thinking blocks
|
||||
logInfo(
|
||||
"Could not determine model capabilities, defaulting to include thinking blocks",
|
||||
error
|
||||
);
|
||||
}
|
||||
|
||||
const streamer = new ThinkBlockStreamer(updateCurrentAiMessage, excludeThinking);
|
||||
|
||||
try {
|
||||
// Tiered lexical retriever doesn't need index check - it builds indexes on demand
|
||||
|
||||
// Require envelope for VaultQA
|
||||
const envelope = userMessage.contextEnvelope;
|
||||
if (!envelope) {
|
||||
throw new Error(
|
||||
"[VaultQA] Context envelope is required but not available. Cannot proceed with VaultQA chain."
|
||||
);
|
||||
}
|
||||
|
||||
// Step 1: Extract L5 (raw user query) from envelope
|
||||
// Tags MUST be extracted from L5 BEFORE condensing to preserve them for tag-aware retrieval
|
||||
const l5User = envelope.layers.find((l) => l.id === "L5_USER");
|
||||
const rawUserQuery = l5User?.text || userMessage.message;
|
||||
|
||||
// Step 2: Extract tags from raw query (BEFORE condensing!)
|
||||
const tags = this.extractTagTerms(rawUserQuery);
|
||||
logInfo("[VaultQA] Extracted tags before condensing:", tags);
|
||||
|
||||
// Step 3: Get chat history from memory (L4)
|
||||
const memory = this.chainManager.memoryManager.getMemory();
|
||||
const memoryVariables = await memory.loadMemoryVariables({});
|
||||
const chatHistory = extractChatHistory(memoryVariables);
|
||||
|
||||
// Step 4: Condense L4 + L5 into standalone question for RAG retrieval
|
||||
// This improves retrieval by incorporating conversation context
|
||||
let standaloneQuestion = rawUserQuery;
|
||||
if (chatHistory.length > 0) {
|
||||
logInfo("[VaultQA] Condensing query with chat history for better retrieval");
|
||||
standaloneQuestion = await getStandaloneQuestion(rawUserQuery, chatHistory);
|
||||
logInfo("[VaultQA] Standalone question:", standaloneQuestion);
|
||||
}
|
||||
|
||||
// Step 5: Create retriever based on semantic search setting
|
||||
const settings = getSettings();
|
||||
|
||||
// Step 5a: Run FilterRetriever for guaranteed title/tag matches
|
||||
const hasTagTerms = tags.length > 0;
|
||||
const filterRetriever = new FilterRetriever(app, {
|
||||
salientTerms: hasTagTerms ? [...tags] : [],
|
||||
maxK: settings.maxSourceChunks,
|
||||
returnAll: hasTagTerms,
|
||||
});
|
||||
const filterDocs = await filterRetriever.getRelevantDocuments(standaloneQuestion);
|
||||
|
||||
// Step 5b: Create main retriever using factory (handles priority: Self-hosted > Semantic > Lexical)
|
||||
// Miyo is only relevant to Plus/agent chains — bypass it for VaultQA.
|
||||
// When Miyo is active, Orama isn't initialized either, so also skip semantic → use lexical.
|
||||
const miyoActive = RetrieverFactory.isMiyoActive();
|
||||
const retrieverResult = await RetrieverFactory.createRetriever(
|
||||
app,
|
||||
{
|
||||
minSimilarityScore: 0.01,
|
||||
maxK: settings.maxSourceChunks,
|
||||
salientTerms: hasTagTerms ? [...tags] : [],
|
||||
tagTerms: tags,
|
||||
returnAll: hasTagTerms,
|
||||
},
|
||||
miyoActive ? { enableMiyo: false, enableSemanticSearchV3: false } : {}
|
||||
);
|
||||
const retriever = retrieverResult.retriever;
|
||||
logInfo(`VaultQA: Using ${retrieverResult.type} retriever - ${retrieverResult.reason}`);
|
||||
|
||||
// Step 5c: Retrieve search results and merge with filter results
|
||||
const searchDocs = await retriever.getRelevantDocuments(standaloneQuestion);
|
||||
const { filterResults, searchResults } = mergeFilterAndSearchResults(filterDocs, searchDocs);
|
||||
// Cap total docs to prevent oversized prompts (filter results prioritized)
|
||||
const merged = [...filterResults, ...searchResults];
|
||||
const retrieverCapReached = merged.length > settings.maxSourceChunks;
|
||||
const retrievedDocs = merged.slice(0, settings.maxSourceChunks);
|
||||
|
||||
// Store retrieved documents for sources
|
||||
this.chainManager.storeRetrieverDocuments(retrievedDocs);
|
||||
|
||||
// Format documents as context with XML tags
|
||||
// Sanitize content to remove pre-existing citation markers
|
||||
|
||||
const context = (
|
||||
retrievedDocs as { metadata?: { title?: string; path?: string }; pageContent?: string }[]
|
||||
)
|
||||
.map((doc) => {
|
||||
const title = doc.metadata?.title || "Untitled";
|
||||
const path = doc.metadata?.path || title;
|
||||
return `<${RETRIEVED_DOCUMENT_TAG}>\n<title>${title}</title>\n<path>${path}</path>\n<content>\n${sanitizeContentForCitations(doc.pageContent ?? "")}\n</content>\n</${RETRIEVED_DOCUMENT_TAG}>`;
|
||||
})
|
||||
.join("\n\n");
|
||||
|
||||
// Step 6: Build messages array with envelope-aware logic
|
||||
const messages: { role: string; content: string | unknown[] }[] = [];
|
||||
const chatModel = this.chainManager.chatModelManager.getChatModel();
|
||||
|
||||
// Prepare RAG context and citation instructions
|
||||
const sourceEntries: SourceCatalogEntry[] = (
|
||||
retrievedDocs as { metadata?: { title?: string; path?: string } }[]
|
||||
)
|
||||
.slice(0, Math.max(5, Math.min(20, retrievedDocs.length)))
|
||||
.map((d) => ({
|
||||
title: d.metadata?.title || d.metadata?.path || "Untitled",
|
||||
path: d.metadata?.path || d.metadata?.title || "",
|
||||
}));
|
||||
const sourceCatalog = formatSourceCatalog(sourceEntries).join("\n");
|
||||
|
||||
const capNotice = retrieverCapReached
|
||||
? `\n\nIMPORTANT: The retrieval limit of ${settings.maxSourceChunks} documents was reached. ${merged.length - settings.maxSourceChunks} additional matching documents were omitted. Inform the user: "Note: The retrieval cap was reached — some matching documents were not included. Upgrade to Copilot Plus for more complete answers."`
|
||||
: "";
|
||||
|
||||
const qaInstructions =
|
||||
"\n\nAnswer the question based only on the following context:\n" +
|
||||
context +
|
||||
getQACitationInstructions(sourceCatalog, settings.enableInlineCitations) +
|
||||
capNotice;
|
||||
|
||||
// Build messages using envelope-based context construction
|
||||
logInfo("[VaultQA] Using envelope-based context construction with LayerToMessagesConverter");
|
||||
|
||||
// Use LayerToMessagesConverter to get base messages with L1+L2 system, L3+L5 user
|
||||
// This ensures smart referencing and L2 Context Library are preserved
|
||||
const baseMessages = LayerToMessagesConverter.convert(envelope, {
|
||||
includeSystemMessage: true,
|
||||
mergeUserContent: true,
|
||||
debug: false,
|
||||
});
|
||||
|
||||
// Add system message (L1 + L2 Context Library only - no RAG)
|
||||
const systemMessage = baseMessages.find((m) => m.role === "system");
|
||||
if (systemMessage) {
|
||||
messages.push({
|
||||
role: getMessageRole(chatModel),
|
||||
content: systemMessage.content,
|
||||
});
|
||||
}
|
||||
|
||||
// Insert L4 (chat history) between system and user
|
||||
await loadAndAddChatHistory(memory, messages);
|
||||
|
||||
// Add user message with RAG prepended
|
||||
// User message now contains: RAG results + citations + L3 smart references + L5
|
||||
// LayerToMessagesConverter already handles smart referencing:
|
||||
// - Items in L2 → referenced by ID
|
||||
// - Items NOT in L2 → full content
|
||||
const userMessageContent = baseMessages.find((m) => m.role === "user");
|
||||
if (userMessageContent) {
|
||||
// Prepend RAG results and citations to user content with proper separator
|
||||
const enhancedUserContent = qaInstructions + "\n\n" + userMessageContent.content;
|
||||
|
||||
// Handle multimodal content if present
|
||||
if (userMessage.content && Array.isArray(userMessage.content)) {
|
||||
const updatedContent = userMessage.content.map(
|
||||
(item: { type?: string }): { type?: string; [key: string]: unknown } => {
|
||||
if (item.type === "text") {
|
||||
return { ...item, text: enhancedUserContent };
|
||||
}
|
||||
return { ...item };
|
||||
}
|
||||
);
|
||||
messages.push({
|
||||
role: "user",
|
||||
content: updatedContent,
|
||||
});
|
||||
} else {
|
||||
messages.push({
|
||||
role: "user",
|
||||
content: enhancedUserContent,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Record the payload for debugging (includes layered view if envelope available)
|
||||
const modelName = (chatModel as { modelName?: string } | undefined)?.modelName;
|
||||
recordPromptPayload({
|
||||
messages,
|
||||
modelName,
|
||||
contextEnvelope: userMessage.contextEnvelope,
|
||||
});
|
||||
|
||||
logInfo("Final Request to AI:\n", messages);
|
||||
|
||||
// Stream with abort signal
|
||||
const chatStream = await withSuppressedTokenWarnings(() =>
|
||||
this.chainManager.chatModelManager.getChatModel().stream(
|
||||
// ProviderMessage[] format matches what getChatModel().stream() accepts at runtime
|
||||
messages as never,
|
||||
{
|
||||
signal: abortController.signal,
|
||||
}
|
||||
)
|
||||
);
|
||||
|
||||
for await (const chunk of chatStream) {
|
||||
if (abortController.signal.aborted) {
|
||||
logInfo("VaultQA stream iteration aborted", { reason: abortController.signal.reason });
|
||||
break;
|
||||
}
|
||||
streamer.processChunk(chunk as Parameters<typeof streamer.processChunk>[0]);
|
||||
}
|
||||
} catch (error: unknown) {
|
||||
// Check if the error is due to abort signal
|
||||
if ((error as { name?: string }).name === "AbortError" || abortController.signal.aborted) {
|
||||
logInfo("VaultQA stream aborted by user", { reason: abortController.signal.reason });
|
||||
// Don't show error message for user-initiated aborts
|
||||
} else {
|
||||
await this.handleError(
|
||||
error,
|
||||
streamer.processErrorChunk.bind(streamer) as (message: string) => void
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Always get the response, even if partial
|
||||
const result = streamer.close();
|
||||
|
||||
const responseMetadata = {
|
||||
wasTruncated: result.wasTruncated,
|
||||
tokenUsage: result.tokenUsage ?? undefined,
|
||||
};
|
||||
|
||||
// Only skip saving if it's a new chat (clearing everything)
|
||||
if (abortController.signal.aborted && abortController.signal.reason === ABORT_REASON.NEW_CHAT) {
|
||||
updateCurrentAiMessage("");
|
||||
return "";
|
||||
}
|
||||
|
||||
// Add sources to the response
|
||||
const fullAIResponse = this.addSourcestoResponse(result.content);
|
||||
|
||||
await this.handleResponse(
|
||||
fullAIResponse,
|
||||
userMessage,
|
||||
abortController,
|
||||
addMessage,
|
||||
updateCurrentAiMessage,
|
||||
undefined,
|
||||
undefined,
|
||||
responseMetadata
|
||||
);
|
||||
|
||||
return fullAIResponse;
|
||||
}
|
||||
|
||||
private addSourcestoResponse(response: string): string {
|
||||
const settings = getSettings();
|
||||
|
||||
// Only add sources if the AI actually cited them (has inline citations like [^1], [^2])
|
||||
// Don't add fallback sources if there are no citations in the response
|
||||
if (!hasInlineCitations(response)) {
|
||||
return response;
|
||||
}
|
||||
|
||||
const retrievedDocs = this.chainManager.getRetrievedDocuments();
|
||||
const sources = extractUniqueTitlesFromDocs(retrievedDocs).map((title) => ({ title }));
|
||||
|
||||
return addFallbackSources(response, sources, settings.enableInlineCitations);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts hash-prefixed tags from the current query so Vault QA can trigger tag-aware retrieval.
|
||||
*/
|
||||
private extractTagTerms(query: string): string[] {
|
||||
return extractTagsFromQuery(query);
|
||||
}
|
||||
}
|
||||
7
src/LLMProviders/chainRunner/index.ts
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
// Main exports for chain runners
|
||||
export type { ChainRunner } from "./BaseChainRunner";
|
||||
export { LLMChainRunner } from "./LLMChainRunner";
|
||||
export { VaultQAChainRunner } from "./VaultQAChainRunner";
|
||||
export { CopilotPlusChainRunner } from "./CopilotPlusChainRunner";
|
||||
export { ProjectChainRunner } from "./ProjectChainRunner";
|
||||
export { AutonomousAgentChainRunner } from "./AutonomousAgentChainRunner";
|
||||
232
src/LLMProviders/chainRunner/utils/ActionBlockStreamer.test.ts
Normal file
|
|
@ -0,0 +1,232 @@
|
|||
import { ActionBlockStreamer } from "./ActionBlockStreamer";
|
||||
import { ToolManager } from "@/tools/toolManager";
|
||||
import { ToolResultFormatter } from "@/tools/ToolResultFormatter";
|
||||
|
||||
// Mock the ToolManager and ToolResultFormatter
|
||||
jest.mock("@/tools/toolManager");
|
||||
jest.mock("@/tools/ToolResultFormatter");
|
||||
|
||||
const MockedToolManager = ToolManager as jest.Mocked<typeof ToolManager>;
|
||||
const MockedToolResultFormatter = ToolResultFormatter as jest.Mocked<typeof ToolResultFormatter>;
|
||||
|
||||
describe("ActionBlockStreamer", () => {
|
||||
let writeFileTool: unknown;
|
||||
let streamer: ActionBlockStreamer;
|
||||
|
||||
beforeEach(() => {
|
||||
writeFileTool = { name: "writeFile" };
|
||||
MockedToolManager.callTool.mockClear();
|
||||
|
||||
// Mock ToolResultFormatter to return the raw result without "File change result: " prefix
|
||||
MockedToolResultFormatter.format = jest.fn((_toolName, result) => result);
|
||||
|
||||
streamer = new ActionBlockStreamer(MockedToolManager, writeFileTool);
|
||||
});
|
||||
|
||||
// Helper function to process chunks and collect results
|
||||
async function processChunks(chunks: { content: string | null }[]): Promise<unknown[]> {
|
||||
const outputContents: unknown[] = [];
|
||||
for (const chunk of chunks) {
|
||||
for await (const result of streamer.processChunk(chunk)) {
|
||||
// Always push the content, even if it's null, undefined, or empty string
|
||||
outputContents.push(result.content);
|
||||
}
|
||||
}
|
||||
return outputContents;
|
||||
}
|
||||
|
||||
it("should pass through chunks without writeFile tags unchanged", async () => {
|
||||
const chunks = [{ content: "Hello " }, { content: "world, this is " }, { content: "a test." }];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
// All chunks should be yielded as-is
|
||||
expect(output).toEqual(["Hello ", "world, this is ", "a test."]);
|
||||
|
||||
// No tool calls should be made
|
||||
expect(MockedToolManager.callTool).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("should handle a complete writeFile block in a single chunk", async () => {
|
||||
MockedToolManager.callTool.mockResolvedValue("File written successfully.");
|
||||
const chunks = [
|
||||
{
|
||||
content:
|
||||
"Some text before <writeFile><path>file.txt</path><content>content</content></writeFile> and some text after.",
|
||||
},
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
// Should yield original chunk plus tool result
|
||||
expect(output).toEqual([
|
||||
"Some text before <writeFile><path>file.txt</path><content>content</content></writeFile> and some text after.",
|
||||
"\nFile written successfully.\n",
|
||||
]);
|
||||
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "file.txt",
|
||||
content: "content",
|
||||
});
|
||||
});
|
||||
|
||||
it("should handle a complete xml-wrapped writeFile block", async () => {
|
||||
MockedToolManager.callTool.mockResolvedValue("XML file written.");
|
||||
const chunks = [
|
||||
{
|
||||
content:
|
||||
"```xml\n<writeFile><path>file.xml</path><content>xml content</content></writeFile>\n```",
|
||||
},
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
expect(output).toEqual([
|
||||
"```xml\n<writeFile><path>file.xml</path><content>xml content</content></writeFile>\n```",
|
||||
"\nXML file written.\n",
|
||||
]);
|
||||
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "file.xml",
|
||||
content: "xml content",
|
||||
});
|
||||
});
|
||||
|
||||
it("should handle a writeFile block split across multiple chunks", async () => {
|
||||
MockedToolManager.callTool.mockResolvedValue("Split file written.");
|
||||
const chunks = [
|
||||
{ content: "Here is a file <writeFile><path>split.txt</path>" },
|
||||
{ content: "<content>split content</content>" },
|
||||
{ content: "</writeFile> That was it." },
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
// All chunks should be yielded as-is, plus tool result when complete block is detected
|
||||
expect(output).toEqual([
|
||||
"Here is a file <writeFile><path>split.txt</path>",
|
||||
"<content>split content</content>",
|
||||
"</writeFile> That was it.",
|
||||
"\nSplit file written.\n",
|
||||
]);
|
||||
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "split.txt",
|
||||
content: "split content",
|
||||
});
|
||||
});
|
||||
|
||||
it("should handle multiple writeFile blocks in the stream", async () => {
|
||||
MockedToolManager.callTool
|
||||
.mockResolvedValueOnce("File 1 written.")
|
||||
.mockResolvedValueOnce("File 2 written.");
|
||||
const chunks = [
|
||||
{
|
||||
content:
|
||||
"<writeFile><path>f1.txt</path><content>c1</content></writeFile>Some text<writeFile><path>f2.txt</path><content>c2</content></writeFile>",
|
||||
},
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
// Should yield original chunk plus both tool results
|
||||
expect(output).toEqual([
|
||||
"<writeFile><path>f1.txt</path><content>c1</content></writeFile>Some text<writeFile><path>f2.txt</path><content>c2</content></writeFile>",
|
||||
"\nFile 1 written.\n",
|
||||
"\nFile 2 written.\n",
|
||||
]);
|
||||
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledTimes(2);
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "f1.txt",
|
||||
content: "c1",
|
||||
});
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "f2.txt",
|
||||
content: "c2",
|
||||
});
|
||||
});
|
||||
|
||||
it("should handle unclosed tags without calling tools", async () => {
|
||||
const chunks = [
|
||||
{ content: "Starting... <writeFile><path>unclosed.txt</path>" },
|
||||
{ content: "<content>this will not be closed" },
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
// Should yield all chunks as-is
|
||||
expect(output).toEqual([
|
||||
"Starting... <writeFile><path>unclosed.txt</path>",
|
||||
"<content>this will not be closed",
|
||||
]);
|
||||
|
||||
// No tool calls should be made for incomplete blocks
|
||||
expect(MockedToolManager.callTool).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("should handle tool call errors gracefully", async () => {
|
||||
MockedToolManager.callTool.mockRejectedValue(new Error("Tool error"));
|
||||
const chunks = [
|
||||
{
|
||||
content: "<writeFile><path>error.txt</path><content>content</content></writeFile>",
|
||||
},
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
expect(output).toEqual([
|
||||
"<writeFile><path>error.txt</path><content>content</content></writeFile>",
|
||||
"\nError: Tool error\n",
|
||||
]);
|
||||
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "error.txt",
|
||||
content: "content",
|
||||
});
|
||||
});
|
||||
|
||||
it("should handle chunks with different content types", async () => {
|
||||
const chunks: { content: string | null }[] = [
|
||||
{ content: "Hello" },
|
||||
{ content: null },
|
||||
{ content: "" },
|
||||
{ content: " World" },
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
// Should yield all chunks as-is, null content is yielded but not added to buffer
|
||||
expect(output).toEqual(["Hello", null, "", " World"]);
|
||||
});
|
||||
|
||||
it("should handle whitespace in path and content", async () => {
|
||||
MockedToolManager.callTool.mockResolvedValue("Whitespace handled.");
|
||||
const chunks = [
|
||||
{
|
||||
content:
|
||||
"<writeFile><path> spaced.txt </path><content> content with spaces </content></writeFile>",
|
||||
},
|
||||
];
|
||||
await processChunks(chunks);
|
||||
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "spaced.txt",
|
||||
content: "content with spaces",
|
||||
});
|
||||
});
|
||||
|
||||
it("should handle malformed blocks gracefully", async () => {
|
||||
MockedToolManager.callTool.mockResolvedValue("Malformed handled.");
|
||||
const chunks = [
|
||||
{
|
||||
content: "<writeFile><path>missing-content.txt</path></writeFile>",
|
||||
},
|
||||
];
|
||||
const output = await processChunks(chunks);
|
||||
|
||||
// Should yield chunk as-is plus tool result
|
||||
expect(output).toEqual([
|
||||
"<writeFile><path>missing-content.txt</path></writeFile>",
|
||||
"\nMalformed handled.\n",
|
||||
]);
|
||||
|
||||
// Tool should be called with undefined content
|
||||
expect(MockedToolManager.callTool).toHaveBeenCalledWith(writeFileTool, {
|
||||
path: "missing-content.txt",
|
||||
content: undefined,
|
||||
});
|
||||
});
|
||||
});
|
||||
95
src/LLMProviders/chainRunner/utils/ActionBlockStreamer.ts
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
import { ToolManager } from "@/tools/toolManager";
|
||||
import { ToolResultFormatter } from "@/tools/ToolResultFormatter";
|
||||
|
||||
/**
|
||||
* ActionBlockStreamer processes streaming chunks to detect and handle writeFile blocks.
|
||||
*
|
||||
* 1. Accumulates chunks in a buffer
|
||||
* 2. Detects complete writeFile blocks
|
||||
* 3. Calls the writeFile tool when a complete block is found
|
||||
* 4. Returns chunks as-is otherwise
|
||||
*/
|
||||
export class ActionBlockStreamer {
|
||||
private buffer = "";
|
||||
|
||||
constructor(
|
||||
private toolManager: typeof ToolManager,
|
||||
private writeFileTool: unknown
|
||||
) {}
|
||||
|
||||
private findCompleteBlock(str: string) {
|
||||
// Regex for both formats
|
||||
const regex = /<writeFile>[\s\S]*?<\/writeFile>/;
|
||||
const match = str.match(regex);
|
||||
|
||||
if (!match || match.index === undefined) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return {
|
||||
block: match[0],
|
||||
endIdx: match.index + match[0].length,
|
||||
};
|
||||
}
|
||||
|
||||
async *processChunk(
|
||||
chunk: Record<string, unknown>
|
||||
): AsyncGenerator<Record<string, unknown>, void, unknown> {
|
||||
// Handle different chunk formats
|
||||
let chunkContent = "";
|
||||
|
||||
// Handle Claude thinking model array-based content
|
||||
if (Array.isArray(chunk.content)) {
|
||||
for (const item of chunk.content as Array<{ type?: string; text?: unknown }>) {
|
||||
if (item.type === "text" && item.text != null) {
|
||||
chunkContent += typeof item.text === "string" ? item.text : "";
|
||||
}
|
||||
}
|
||||
}
|
||||
// Handle standard string content
|
||||
else if (chunk.content != null) {
|
||||
chunkContent = typeof chunk.content === "string" ? chunk.content : "";
|
||||
}
|
||||
|
||||
// Add to buffer
|
||||
if (chunkContent) {
|
||||
this.buffer += chunkContent;
|
||||
}
|
||||
|
||||
// Yield the original chunk as-is
|
||||
yield chunk;
|
||||
|
||||
// Process all complete blocks in the buffer
|
||||
let blockInfo = this.findCompleteBlock(this.buffer);
|
||||
|
||||
while (blockInfo) {
|
||||
const { block, endIdx } = blockInfo;
|
||||
|
||||
// Extract content from the block
|
||||
const pathMatch = block.match(/<path>([\s\S]*?)<\/path>/);
|
||||
const contentMatch = block.match(/<content>([\s\S]*?)<\/content>/);
|
||||
const filePath = pathMatch ? pathMatch[1].trim() : undefined;
|
||||
const fileContent = contentMatch ? contentMatch[1].trim() : undefined;
|
||||
|
||||
// Call the tool
|
||||
try {
|
||||
const result = await this.toolManager.callTool(this.writeFileTool, {
|
||||
path: filePath,
|
||||
content: fileContent,
|
||||
});
|
||||
|
||||
// Format tool result using ToolResultFormatter for consistency with agent mode
|
||||
const formattedResult = ToolResultFormatter.format("writeFile", result as string);
|
||||
yield { ...chunk, content: `\n${formattedResult}\n` };
|
||||
} catch (err: unknown) {
|
||||
yield { ...chunk, content: `\nError: ${(err as Error)?.message ?? String(err)}\n` };
|
||||
}
|
||||
|
||||
// Remove processed block from buffer
|
||||
this.buffer = this.buffer.substring(endIdx);
|
||||
|
||||
// Check for another complete block in the remaining buffer
|
||||
blockInfo = this.findCompleteBlock(this.buffer);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,93 @@
|
|||
import { summarizeToolCall, summarizeToolResult } from "./AgentReasoningState";
|
||||
|
||||
describe("AgentReasoningState tool summaries", () => {
|
||||
test("summarizeToolCall has daily/random CLI specific wording", () => {
|
||||
expect(summarizeToolCall("obsidianDailyNote", { command: "daily:read" })).toBe(
|
||||
"Reading today's daily note"
|
||||
);
|
||||
expect(summarizeToolCall("obsidianDailyNote", { command: "daily:read", vault: "Work" })).toBe(
|
||||
`Reading today's daily note from "Work"`
|
||||
);
|
||||
expect(summarizeToolCall("obsidianDailyNote", { command: "daily:path" })).toBe(
|
||||
"Getting daily note path"
|
||||
);
|
||||
|
||||
expect(summarizeToolCall("obsidianRandomRead")).toBe("Reading a random note");
|
||||
expect(summarizeToolCall("obsidianRandomRead", { vault: "Personal" })).toBe(
|
||||
`Reading a random note from "Personal"`
|
||||
);
|
||||
});
|
||||
|
||||
test("summarizeToolResult has daily/random CLI specific wording", () => {
|
||||
expect(
|
||||
summarizeToolResult("obsidianDailyNote", { success: true }, undefined, {
|
||||
command: "daily:read",
|
||||
vault: "Work",
|
||||
})
|
||||
).toBe(`Loaded today's daily note from "Work"`);
|
||||
expect(
|
||||
summarizeToolResult("obsidianDailyNote", { success: true }, undefined, {
|
||||
command: "daily:read",
|
||||
})
|
||||
).toBe("Loaded today's daily note");
|
||||
expect(
|
||||
summarizeToolResult("obsidianDailyNote", { success: true }, undefined, {
|
||||
command: "daily:path",
|
||||
vault: "Work",
|
||||
})
|
||||
).toBe(`Got daily note path from "Work"`);
|
||||
|
||||
expect(
|
||||
summarizeToolResult("obsidianRandomRead", { success: true }, undefined, { vault: "Personal" })
|
||||
).toBe(`Loaded a random note from "Personal"`);
|
||||
expect(summarizeToolResult("obsidianRandomRead", { success: true })).toBe(
|
||||
"Loaded a random note"
|
||||
);
|
||||
});
|
||||
|
||||
test("summarizeToolResult failure path reuses CLI call summary", () => {
|
||||
expect(
|
||||
summarizeToolResult("obsidianRandomRead", { success: false }, undefined, { vault: "VaultA" })
|
||||
).toBe(`Reading a random note from "VaultA" failed`);
|
||||
});
|
||||
|
||||
test("summarizeToolCall has properties/tasks/links CLI specific wording", () => {
|
||||
expect(summarizeToolCall("obsidianProperties", { command: "properties" })).toBe(
|
||||
"Listing vault properties"
|
||||
);
|
||||
expect(
|
||||
summarizeToolCall("obsidianProperties", { command: "property:read", name: "tags" })
|
||||
).toBe(`Reading property "tags"`);
|
||||
expect(summarizeToolCall("obsidianTasks", { command: "tasks" })).toBe("Listing vault tasks");
|
||||
expect(summarizeToolCall("obsidianLinks", { command: "backlinks" })).toBe("Listing backlinks");
|
||||
expect(summarizeToolCall("obsidianLinks", { command: "orphans" })).toBe(
|
||||
"Listing orphaned notes"
|
||||
);
|
||||
expect(summarizeToolCall("obsidianLinks", { command: "unresolved" })).toBe(
|
||||
"Listing unresolved links"
|
||||
);
|
||||
});
|
||||
|
||||
test("summarizeToolResult has properties/tasks/links CLI specific wording", () => {
|
||||
expect(
|
||||
summarizeToolResult("obsidianProperties", { success: true }, undefined, {
|
||||
command: "properties",
|
||||
})
|
||||
).toBe("Listed vault properties");
|
||||
expect(
|
||||
summarizeToolResult("obsidianProperties", { success: true }, undefined, {
|
||||
command: "property:read",
|
||||
name: "tags",
|
||||
})
|
||||
).toBe(`Read property "tags"`);
|
||||
expect(
|
||||
summarizeToolResult("obsidianTasks", { success: true }, undefined, { command: "tasks" })
|
||||
).toBe("Listed vault tasks");
|
||||
expect(
|
||||
summarizeToolResult("obsidianLinks", { success: true }, undefined, { command: "backlinks" })
|
||||
).toBe("Listed backlinks");
|
||||
expect(
|
||||
summarizeToolResult("obsidianLinks", { success: true }, undefined, { command: "orphans" })
|
||||
).toBe("Listed orphaned notes");
|
||||
});
|
||||
});
|
||||
471
src/LLMProviders/chainRunner/utils/AgentReasoningState.ts
Normal file
|
|
@ -0,0 +1,471 @@
|
|||
/**
|
||||
* Agent Reasoning Block State Management
|
||||
*
|
||||
* This module provides state management for the Agent Reasoning Block UI component,
|
||||
* which replaces the old tool call banner with a more informative reasoning display.
|
||||
*/
|
||||
|
||||
/**
|
||||
* Represents a single reasoning step in the agent loop
|
||||
*/
|
||||
export interface ReasoningStep {
|
||||
timestamp: number;
|
||||
summary: string;
|
||||
toolName?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Status of the reasoning block
|
||||
* - idle: No agent activity
|
||||
* - reasoning: Agent is actively processing/executing tools
|
||||
* - collapsed: Reasoning complete, block is collapsed
|
||||
* - complete: Response complete, block can be expanded
|
||||
*/
|
||||
export type ReasoningStatus = "idle" | "reasoning" | "collapsed" | "complete";
|
||||
|
||||
/**
|
||||
* Full state for the Agent Reasoning Block
|
||||
*/
|
||||
export interface AgentReasoningState {
|
||||
status: ReasoningStatus;
|
||||
startTime: number | null;
|
||||
elapsedSeconds: number;
|
||||
steps: ReasoningStep[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates the initial reasoning state
|
||||
*/
|
||||
export function createInitialReasoningState(): AgentReasoningState {
|
||||
return {
|
||||
status: "idle",
|
||||
startTime: null,
|
||||
elapsedSeconds: 0,
|
||||
steps: [],
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Data structure for serialized reasoning block (embedded in message)
|
||||
*/
|
||||
export interface SerializedReasoningData {
|
||||
elapsed: number;
|
||||
steps: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Serialize reasoning state to a marker format for embedding in messages.
|
||||
* Format: <!--AGENT_REASONING:status:elapsedSeconds:["step1","step2"]-->
|
||||
*
|
||||
* @param state - The reasoning state to serialize
|
||||
* @returns Marker string to embed in message
|
||||
*/
|
||||
export function serializeReasoningBlock(state: AgentReasoningState): string {
|
||||
if (state.status === "idle") {
|
||||
return "";
|
||||
}
|
||||
|
||||
const stepsJson = JSON.stringify(state.steps.map((s) => s.summary));
|
||||
return `<!--AGENT_REASONING:${state.status}:${state.elapsedSeconds}:${stepsJson}-->`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parsed reasoning data from a marker
|
||||
*/
|
||||
export interface ParsedReasoningBlock {
|
||||
hasReasoning: boolean;
|
||||
status: ReasoningStatus;
|
||||
elapsedSeconds: number;
|
||||
steps: string[];
|
||||
contentAfter: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse reasoning block marker from message content.
|
||||
*
|
||||
* @param content - Message content that may contain reasoning marker
|
||||
* @returns Parsed reasoning data or null if no marker found
|
||||
*/
|
||||
export function parseReasoningBlock(content: string): ParsedReasoningBlock | null {
|
||||
const match = content.match(/<!--AGENT_REASONING:(\w+):(\d+):(.+?)-->/);
|
||||
if (!match) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const [fullMatch, status, elapsed, stepsJson] = match;
|
||||
|
||||
let steps: string[] = [];
|
||||
try {
|
||||
steps = JSON.parse(stepsJson) as string[];
|
||||
} catch {
|
||||
// Invalid JSON, return empty steps
|
||||
steps = [];
|
||||
}
|
||||
|
||||
return {
|
||||
hasReasoning: true,
|
||||
status: status as ReasoningStatus,
|
||||
elapsedSeconds: parseInt(elapsed, 10),
|
||||
steps,
|
||||
contentAfter: content.replace(fullMatch, "").trim(),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Query expansion info for localSearch.
|
||||
* Contains both the individual expansion components and a combined list of all recall terms.
|
||||
*/
|
||||
export interface QueryExpansionInfo {
|
||||
originalQuery: string;
|
||||
salientTerms: string[]; // Terms from original query (used for ranking)
|
||||
expandedQueries: string[]; // Alternative phrasings (used for recall)
|
||||
recallTerms: string[]; // All terms combined that were used for recall
|
||||
}
|
||||
|
||||
/**
|
||||
* Source info for localSearch results
|
||||
*/
|
||||
export interface LocalSearchSourceInfo {
|
||||
titles: string[];
|
||||
count: number;
|
||||
queryExpansion?: QueryExpansionInfo;
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a human-readable summary for a tool result.
|
||||
*
|
||||
* @param toolName - Name of the tool that was executed
|
||||
* @param result - Result from the tool execution
|
||||
* @param sourceInfo - Optional source info for localSearch results
|
||||
* @param args - Optional original tool call arguments for context
|
||||
* @returns Human-readable summary string
|
||||
*/
|
||||
export function summarizeToolResult(
|
||||
toolName: string,
|
||||
result: { success: boolean; result?: string },
|
||||
sourceInfo?: LocalSearchSourceInfo,
|
||||
args?: Record<string, unknown>
|
||||
): string {
|
||||
if (!result.success) {
|
||||
// Reuse the human-friendly call summary (e.g., "Searching notes") → "Searching notes failed"
|
||||
return `${summarizeToolCall(toolName, args)} failed`;
|
||||
}
|
||||
|
||||
switch (toolName) {
|
||||
case "localSearch": {
|
||||
if (sourceInfo && sourceInfo.count > 0) {
|
||||
// Show just the count and first few note titles (terms are shown in tool call summary)
|
||||
const titleList = sourceInfo.titles.slice(0, 3);
|
||||
const remaining = sourceInfo.count - titleList.length;
|
||||
let result = `Found ${sourceInfo.count} note${sourceInfo.count !== 1 ? "s" : ""}: ${titleList.join(", ")}`;
|
||||
if (remaining > 0) {
|
||||
result += ` +${remaining} more`;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
return "No matching notes found";
|
||||
}
|
||||
case "webSearch":
|
||||
return "Retrieved web search results";
|
||||
case "getTimeRangeMs":
|
||||
return "Calculated time range";
|
||||
case "readFile":
|
||||
return "Read file content";
|
||||
case "readNote": {
|
||||
const notePath = args?.notePath as string | undefined;
|
||||
if (notePath) {
|
||||
const noteTitle = notePath.split("/").pop()?.replace(/\.md$/i, "") || notePath;
|
||||
return `Read "${noteTitle}"`;
|
||||
}
|
||||
return "Read note content";
|
||||
}
|
||||
case "createNote":
|
||||
return "Created new note";
|
||||
case "appendToNote":
|
||||
return "Appended to note";
|
||||
case "editNote":
|
||||
return "Edited note";
|
||||
case "deleteNote":
|
||||
return "Deleted note";
|
||||
case "youtubeTranscript":
|
||||
case "youtubeTranscription":
|
||||
return "Fetched video transcript";
|
||||
case "fetchUrl":
|
||||
return "Fetched URL content";
|
||||
case "getFileTree":
|
||||
return "Retrieved vault file tree";
|
||||
case "getTagList":
|
||||
return "Retrieved vault tags";
|
||||
case "getCurrentTime":
|
||||
return "Got current time";
|
||||
case "getTimeInfoByEpoch":
|
||||
return "Converted timestamp";
|
||||
case "convertTimeBetweenTimezones":
|
||||
return "Converted timezone";
|
||||
case "obsidianDailyNote": {
|
||||
const command = args?.command as string | undefined;
|
||||
const vault = args?.vault as string | undefined;
|
||||
const vaultSuffix = vault && vault.trim().length > 0 ? ` from "${vault}"` : "";
|
||||
if (command === "daily:path") return `Got daily note path${vaultSuffix}`;
|
||||
return `Loaded today's daily note${vaultSuffix}`;
|
||||
}
|
||||
case "obsidianRandomRead": {
|
||||
const vault = args?.vault as string | undefined;
|
||||
if (vault && vault.trim().length > 0) {
|
||||
return `Loaded a random note from "${vault}"`;
|
||||
}
|
||||
return "Loaded a random note";
|
||||
}
|
||||
case "obsidianProperties": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "property:read") {
|
||||
const name = args?.name as string | undefined;
|
||||
return name ? `Read property "${name}"` : "Read property";
|
||||
}
|
||||
return "Listed vault properties";
|
||||
}
|
||||
case "obsidianTasks":
|
||||
return "Listed vault tasks";
|
||||
case "obsidianLinks": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "backlinks") return "Listed backlinks";
|
||||
if (command === "links") return "Listed outgoing links";
|
||||
if (command === "orphans") return "Listed orphaned notes";
|
||||
if (command === "unresolved") return "Listed unresolved links";
|
||||
return "Queried link graph";
|
||||
}
|
||||
case "obsidianTemplates": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "template:read") {
|
||||
const name = args?.name as string | undefined;
|
||||
return name ? `Read template "${name}"` : "Read template";
|
||||
}
|
||||
return "Listed templates";
|
||||
}
|
||||
case "obsidianBases": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "base:views") return "Listed base views";
|
||||
if (command === "base:query") return "Queried base data";
|
||||
return "Listed bases";
|
||||
}
|
||||
case "indexVault":
|
||||
return "Indexed vault";
|
||||
case "updateMemory":
|
||||
return "Updated memory";
|
||||
case "writeFile":
|
||||
case "editFile": {
|
||||
// Parse the result to check if accepted/rejected
|
||||
const filePath = args?.path as string | undefined;
|
||||
const fileName = filePath ? filePath.split("/").pop() || filePath : "file";
|
||||
|
||||
// Result is JSON string, check for rejected/accepted status
|
||||
const resultStr = result.result || "";
|
||||
if (resultStr.includes('"rejected"') || resultStr.includes("rejected")) {
|
||||
return `Edit rejected for "${fileName}"`;
|
||||
}
|
||||
if (resultStr.includes('"failed"') || resultStr.includes("Error")) {
|
||||
return `Edit failed for "${fileName}"`;
|
||||
}
|
||||
// TODO(@wenzhengjiang): Handle no-op cases (e.g., "File is too small", "Search text not found")
|
||||
// Requires ComposerTools to return structured results instead of plain strings.
|
||||
// See docs/TODO-composer-tool-redesign.md
|
||||
return toolName === "writeFile" ? `Wrote to "${fileName}"` : `Edited "${fileName}"`;
|
||||
}
|
||||
default:
|
||||
return "Done";
|
||||
}
|
||||
}
|
||||
|
||||
// TODO: The `expansion` parameter and QueryExpansionInfo interface are now dead code --
|
||||
// the agent runner no longer pre-expands queries. Clean up the expansion branch below
|
||||
// and the _preExpandedQuery schema field in SearchTools.ts.
|
||||
/**
|
||||
* Generate a summary for when a tool is being called.
|
||||
*
|
||||
* @param toolName - Name of the tool being called
|
||||
* @param args - Arguments being passed to the tool
|
||||
* @param expansion - Optional pre-expanded query data for localSearch
|
||||
* @returns Human-readable summary string
|
||||
*/
|
||||
export function summarizeToolCall(
|
||||
toolName: string,
|
||||
args?: Record<string, unknown>,
|
||||
expansion?: QueryExpansionInfo
|
||||
): string {
|
||||
switch (toolName) {
|
||||
case "localSearch": {
|
||||
// If we have pre-expanded terms, show all recall terms
|
||||
if (expansion && expansion.recallTerms && expansion.recallTerms.length > 0) {
|
||||
// Filter to valid strings only, excluding "[object Object]" artifacts
|
||||
const validTerms = expansion.recallTerms.filter(
|
||||
(t): t is string =>
|
||||
typeof t === "string" &&
|
||||
t.trim().length > 0 &&
|
||||
!t.includes("[object ") &&
|
||||
t !== "[object Object]"
|
||||
);
|
||||
if (validTerms.length > 0) {
|
||||
const terms = validTerms
|
||||
.slice(0, 6)
|
||||
.map((t) => `"${t}"`)
|
||||
.join(", ");
|
||||
const moreCount = validTerms.length - 6;
|
||||
const termsSuffix = moreCount > 0 ? ` +${moreCount} more` : "";
|
||||
return `Searching notes for ${terms}${termsSuffix}`;
|
||||
}
|
||||
}
|
||||
// Fallback to query if no expansion available
|
||||
const query = args?.query as string | undefined;
|
||||
if (query) {
|
||||
const truncatedQuery = query.length > 50 ? query.slice(0, 50) + "..." : query;
|
||||
return `Searching notes for "${truncatedQuery}"`;
|
||||
}
|
||||
return "Searching notes";
|
||||
}
|
||||
case "webSearch": {
|
||||
const query = args?.query as string | undefined;
|
||||
if (query) {
|
||||
const truncatedQuery = query.length > 30 ? query.slice(0, 30) + "..." : query;
|
||||
return `Searching web for "${truncatedQuery}"`;
|
||||
}
|
||||
return "Searching the web";
|
||||
}
|
||||
case "getTimeRangeMs":
|
||||
return "Calculating time range";
|
||||
case "readFile": {
|
||||
const path = args?.path as string | undefined;
|
||||
if (path) {
|
||||
const fileName = path.split("/").pop() || path;
|
||||
return `Reading "${fileName}"`;
|
||||
}
|
||||
return "Reading file";
|
||||
}
|
||||
case "readNote": {
|
||||
const notePath = args?.notePath as string | undefined;
|
||||
if (notePath) {
|
||||
// Extract note title from path (remove .md extension and get last segment)
|
||||
const noteTitle = notePath.split("/").pop()?.replace(/\.md$/i, "") || notePath;
|
||||
return `Reading "${noteTitle}"`;
|
||||
}
|
||||
return "Reading note";
|
||||
}
|
||||
case "createNote":
|
||||
return "Creating new note";
|
||||
case "appendToNote":
|
||||
return "Appending to note";
|
||||
case "editNote":
|
||||
return "Editing note";
|
||||
case "deleteNote":
|
||||
return "Deleting note";
|
||||
case "youtubeTranscript":
|
||||
case "youtubeTranscription":
|
||||
return "Fetching video transcript";
|
||||
case "fetchUrl":
|
||||
return "Fetching URL content";
|
||||
case "getFileTree":
|
||||
return "Browsing vault file tree";
|
||||
case "getTagList":
|
||||
return "Loading vault tags";
|
||||
case "getCurrentTime":
|
||||
return "Getting current time";
|
||||
case "getTimeInfoByEpoch":
|
||||
return "Converting timestamp";
|
||||
case "convertTimeBetweenTimezones":
|
||||
return "Converting timezone";
|
||||
case "obsidianDailyNote": {
|
||||
const command = args?.command as string | undefined;
|
||||
const vault = args?.vault as string | undefined;
|
||||
const vaultSuffix = vault && vault.trim().length > 0 ? ` from "${vault}"` : "";
|
||||
if (command === "daily:path") return `Getting daily note path${vaultSuffix}`;
|
||||
return `Reading today's daily note${vaultSuffix}`;
|
||||
}
|
||||
case "obsidianRandomRead": {
|
||||
const vault = args?.vault as string | undefined;
|
||||
if (vault && vault.trim().length > 0) {
|
||||
return `Reading a random note from "${vault}"`;
|
||||
}
|
||||
return "Reading a random note";
|
||||
}
|
||||
case "obsidianProperties": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "property:read") {
|
||||
const name = args?.name as string | undefined;
|
||||
return name ? `Reading property "${name}"` : "Reading property";
|
||||
}
|
||||
return "Listing vault properties";
|
||||
}
|
||||
case "obsidianTasks":
|
||||
return "Listing vault tasks";
|
||||
case "obsidianLinks": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "backlinks") return "Listing backlinks";
|
||||
if (command === "links") return "Listing outgoing links";
|
||||
if (command === "orphans") return "Listing orphaned notes";
|
||||
if (command === "unresolved") return "Listing unresolved links";
|
||||
return "Querying link graph";
|
||||
}
|
||||
case "obsidianTemplates": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "template:read") {
|
||||
const name = args?.name as string | undefined;
|
||||
return name ? `Reading template "${name}"` : "Reading template";
|
||||
}
|
||||
return "Listing templates";
|
||||
}
|
||||
case "obsidianBases": {
|
||||
const command = args?.command as string | undefined;
|
||||
if (command === "base:views") return "Listing base views";
|
||||
if (command === "base:query") return "Querying base data";
|
||||
return "Listing bases";
|
||||
}
|
||||
case "indexVault":
|
||||
return "Indexing vault";
|
||||
case "updateMemory":
|
||||
return "Saving to memory";
|
||||
case "writeFile": {
|
||||
const filePath = args?.path as string | undefined;
|
||||
if (filePath) {
|
||||
const fileName = filePath.split("/").pop() || filePath;
|
||||
return `Writing to "${fileName}"`;
|
||||
}
|
||||
return "Writing to file";
|
||||
}
|
||||
case "editFile": {
|
||||
const filePath = args?.path as string | undefined;
|
||||
if (filePath) {
|
||||
const fileName = filePath.split("/").pop() || filePath;
|
||||
return `Editing "${fileName}"`;
|
||||
}
|
||||
return "Editing file";
|
||||
}
|
||||
default:
|
||||
return "Processing";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Truncate text to a maximum length, adding ellipsis if needed.
|
||||
*/
|
||||
function truncate(text: string, maxLen: number): string {
|
||||
return text.length > maxLen ? text.slice(0, maxLen - 3) + "..." : text;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the first sentence from model's intermediate reasoning content.
|
||||
* Used to show what the model found/concluded from previous tool calls.
|
||||
*
|
||||
* Uses a simple approach: take first line, truncate if needed.
|
||||
* This avoids edge cases with abbreviations (Dr., U.S.) that confuse regex-based
|
||||
* sentence detection.
|
||||
*
|
||||
* @param content - Model's intermediate content (may contain reasoning about findings)
|
||||
* @returns First line/sentence if found, null otherwise
|
||||
*/
|
||||
export function extractFirstSentence(content: string): string | null {
|
||||
if (!content || content.trim().length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const firstLine = content.trim().split("\n")[0];
|
||||
return truncate(firstLine, 100);
|
||||
}
|
||||
413
src/LLMProviders/chainRunner/utils/ThinkBlockStreamer.test.ts
Normal file
|
|
@ -0,0 +1,413 @@
|
|||
import { ThinkBlockStreamer } from "./ThinkBlockStreamer";
|
||||
|
||||
describe("ThinkBlockStreamer", () => {
|
||||
describe("OpenRouter delta.reasoning format", () => {
|
||||
it("should NOT treat empty reasoning_details array as thinking content", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
// This was the bug: empty reasoning_details array should not trigger thinking mode
|
||||
streamer.processChunk({
|
||||
content: "Regular content",
|
||||
additional_kwargs: {
|
||||
reasoning_details: [],
|
||||
},
|
||||
});
|
||||
|
||||
// Should NOT have <think> tags since reasoning_details is empty
|
||||
expect(currentMessage).toBe("Regular content");
|
||||
expect(currentMessage).not.toContain("<think>");
|
||||
});
|
||||
|
||||
it("should handle delta.reasoning for streaming", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
// First chunk with delta.reasoning
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: "Thinking step 1: ",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Thinking step 1: ");
|
||||
|
||||
// Second chunk with more delta.reasoning
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: "Thinking step 2.",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Thinking step 1: Thinking step 2.");
|
||||
|
||||
// Regular content should close think block
|
||||
streamer.processChunk({
|
||||
content: "Here's the result.",
|
||||
additional_kwargs: {},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe(
|
||||
"\n<think>Thinking step 1: Thinking step 2.</think>Here's the result."
|
||||
);
|
||||
});
|
||||
|
||||
it("should NOT duplicate when both delta.reasoning and reasoning_details are present", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
// First chunk: delta.reasoning with streaming token
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: "Analyzing the ",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Analyzing the ");
|
||||
|
||||
// Second chunk: more delta.reasoning
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: "question carefully.",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Analyzing the question carefully.");
|
||||
|
||||
// Final chunk: reasoning_details with complete transcript (should be IGNORED)
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
reasoning_details: [
|
||||
{
|
||||
text: "Analyzing the question carefully.", // Same content as accumulated delta
|
||||
},
|
||||
],
|
||||
},
|
||||
});
|
||||
|
||||
// Should NOT duplicate - reasoning_details should be ignored since we've seen delta.reasoning
|
||||
expect(currentMessage).toBe("\n<think>Analyzing the question carefully.");
|
||||
expect(currentMessage).not.toContain(
|
||||
"Analyzing the question carefully.Analyzing the question carefully."
|
||||
);
|
||||
|
||||
// Regular content
|
||||
streamer.processChunk({
|
||||
content: "Here's my answer.",
|
||||
additional_kwargs: {},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe(
|
||||
"\n<think>Analyzing the question carefully.</think>Here's my answer."
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe("Claude array-based format", () => {
|
||||
it("should handle Claude's content array with thinking type", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
// Claude format with content array
|
||||
streamer.processChunk({
|
||||
content: [
|
||||
{
|
||||
type: "thinking",
|
||||
thinking: "Let me analyze this...",
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Let me analyze this...");
|
||||
|
||||
// Text content in array
|
||||
streamer.processChunk({
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "Based on my analysis, ",
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Let me analyze this...</think>Based on my analysis, ");
|
||||
});
|
||||
|
||||
it("should guard against undefined thinking content in Claude format", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
// Malformed chunk with undefined thinking
|
||||
streamer.processChunk({
|
||||
content: [
|
||||
{
|
||||
type: "thinking",
|
||||
// thinking property is undefined
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
// Should not crash, and should not add undefined to response
|
||||
expect(currentMessage).toBe("\n<think>");
|
||||
expect(currentMessage).not.toContain("undefined");
|
||||
});
|
||||
});
|
||||
|
||||
describe("Deepseek format", () => {
|
||||
it("should handle Deepseek reasoning_content", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
reasoning_content: "Deepseek is thinking...",
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Deepseek is thinking...");
|
||||
|
||||
streamer.processChunk({
|
||||
content: "The answer is here.",
|
||||
additional_kwargs: {},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Deepseek is thinking...</think>The answer is here.");
|
||||
});
|
||||
|
||||
it("should guard against undefined reasoning_content in Deepseek format", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
// Malformed chunk with undefined reasoning_content
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
reasoning_content: undefined,
|
||||
},
|
||||
});
|
||||
|
||||
// Should not crash, not open think block, and not add undefined
|
||||
expect(currentMessage).toBe("");
|
||||
expect(currentMessage).not.toContain("undefined");
|
||||
expect(currentMessage).not.toContain("<think>");
|
||||
});
|
||||
|
||||
it("should handle streaming Deepseek reasoning_content without premature closure", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
// First chunk with reasoning_content
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
reasoning_content: "Thinking step 1...",
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Thinking step 1...");
|
||||
|
||||
// Second chunk with MORE reasoning_content (streaming)
|
||||
// This should NOT close and reopen the think block
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
reasoning_content: " Step 2...",
|
||||
},
|
||||
});
|
||||
|
||||
// Should be continuous, NOT "</think>\n<think>"
|
||||
expect(currentMessage).toBe("\n<think>Thinking step 1... Step 2...");
|
||||
expect(currentMessage).not.toContain("</think>\n<think>");
|
||||
|
||||
// Third chunk with regular content
|
||||
streamer.processChunk({
|
||||
content: "Final answer.",
|
||||
additional_kwargs: {},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Thinking step 1... Step 2...</think>Final answer.");
|
||||
});
|
||||
});
|
||||
|
||||
describe("excludeThinking option", () => {
|
||||
it("should skip OpenRouter thinking content when excludeThinking is true", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer(
|
||||
(msg) => {
|
||||
currentMessage = msg;
|
||||
},
|
||||
true // excludeThinking = true
|
||||
);
|
||||
|
||||
// Thinking content should be skipped
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: "This should be skipped",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("");
|
||||
|
||||
// Regular content should still be processed
|
||||
streamer.processChunk({
|
||||
content: "This should be included",
|
||||
additional_kwargs: {},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("This should be included");
|
||||
});
|
||||
|
||||
it("should skip Claude thinking content when excludeThinking is true", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer(
|
||||
(msg) => {
|
||||
currentMessage = msg;
|
||||
},
|
||||
true // excludeThinking = true
|
||||
);
|
||||
|
||||
streamer.processChunk({
|
||||
content: [
|
||||
{
|
||||
type: "thinking",
|
||||
thinking: "Claude thinking",
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("");
|
||||
expect(currentMessage).not.toContain("<think>");
|
||||
});
|
||||
});
|
||||
|
||||
describe("close() method", () => {
|
||||
it("should close any open think block at the end", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: "Thinking...",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Thinking...");
|
||||
|
||||
const result = streamer.close();
|
||||
expect(result.content).toBe("\n<think>Thinking...</think>");
|
||||
});
|
||||
|
||||
it("should not add extra closing tag if already closed", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: "Thinking...",
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
streamer.processChunk({
|
||||
content: "Done",
|
||||
additional_kwargs: {},
|
||||
});
|
||||
|
||||
expect(currentMessage).toBe("\n<think>Thinking...</think>Done");
|
||||
|
||||
const result = streamer.close();
|
||||
expect(result.content).toBe("\n<think>Thinking...</think>Done");
|
||||
// Should not have double closing tags
|
||||
expect(result.content.match(/<\/think>/g)?.length).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("mixed content scenarios", () => {
|
||||
it("should handle rapid alternation between thinking and regular content", () => {
|
||||
let currentMessage = "";
|
||||
const streamer = new ThinkBlockStreamer((msg) => {
|
||||
currentMessage = msg;
|
||||
});
|
||||
|
||||
const chunks = [
|
||||
{ thinking: "Think 1", content: "" },
|
||||
{ thinking: "", content: "Text 1" },
|
||||
{ thinking: "Think 2", content: "" },
|
||||
{ thinking: "", content: "Text 2" },
|
||||
{ thinking: "Think 3", content: "" },
|
||||
{ thinking: "", content: "Text 3" },
|
||||
];
|
||||
|
||||
chunks.forEach((chunk) => {
|
||||
if (chunk.thinking) {
|
||||
streamer.processChunk({
|
||||
content: "",
|
||||
additional_kwargs: {
|
||||
delta: {
|
||||
reasoning: chunk.thinking,
|
||||
},
|
||||
},
|
||||
});
|
||||
} else {
|
||||
streamer.processChunk({
|
||||
content: chunk.content,
|
||||
additional_kwargs: {},
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// Should have three separate think blocks
|
||||
const thinkMatches = currentMessage.match(/<think>/g);
|
||||
const thinkCloseMatches = currentMessage.match(/<\/think>/g);
|
||||
expect(thinkMatches?.length).toBe(3);
|
||||
expect(thinkCloseMatches?.length).toBe(3);
|
||||
|
||||
// Each text should be outside think blocks
|
||||
expect(currentMessage).toContain("</think>Text 1");
|
||||
expect(currentMessage).toContain("</think>Text 2");
|
||||
expect(currentMessage).toContain("</think>Text 3");
|
||||
});
|
||||
});
|
||||
});
|
||||