diff --git a/laptop_full_setup.sh b/laptop_full_setup.sh index 4fa43f9..7946db1 100755 --- a/laptop_full_setup.sh +++ b/laptop_full_setup.sh @@ -66,6 +66,9 @@ elif [[ "$TOTAL_VRAM" -ge 14 ]]; then elif [[ "$TOTAL_VRAM" -ge 8 ]]; then CHAT_MODEL="qwen3.5:9b"; CODE_MODEL="qwen3.5:9b" CTX=32768; GPU_TIER="${TOTAL_VRAM}GB VRAM β€” 9B dense, 32K context" +elif [[ "$TOTAL_VRAM" -ge 6 ]]; then + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=32768; GPU_TIER="${TOTAL_VRAM}GB VRAM β€” 4B + KV quant, 32K context (3.5GB free for cache)" elif [[ "$TOTAL_VRAM" -ge 4 ]]; then CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" CTX=16384; GPU_TIER="${TOTAL_VRAM}GB VRAM β€” 4B models, 16K context" @@ -463,23 +466,32 @@ if $INSTALL_AI; then echo "" echo " Origin preference:" echo " 1) Western-only β€” Codestral (Mistral πŸ‡«πŸ‡·) Β· Phi4 (Microsoft πŸ‡ΊπŸ‡Έ) Β· Mistral 7B" - echo " 2) Performance-first β€” Qwen2.5 Β· Qwen2.5-Coder (Chinese, top benchmarks)" - echo " 3) Mixed β€” Western for chat/reasoning, Qwen for coding only" + echo " 2) Performance-first β€” Qwen 3.5 (Feb 2026, top benchmarks, vision+code)" + echo " 3) Mixed β€” Western for chat, Qwen 3.5 for coding" echo " 4) Custom β€” enter model names manually" echo "" - read -rp " Choice [1]: " MODEL_PREF - MODEL_PREF="${MODEL_PREF:-1}" + read -rp " Choice [2]: " MODEL_PREF + MODEL_PREF="${MODEL_PREF:-2}" - # Q4 approximate sizes in GB: 7B=4, 13-15B=9, 22B=13, 32B=19, 70-72B=41 + # Q4_K_M approximate weights in GB: 4B=2.5, 7B=4, 9B=5.5, 14B=9, 22B=13, 27B=17, 35B-MoE=12, 70B=41 # VRAM-based recommendation (soft β€” shown as suggestion only) - if [[ "$VRAM_GB" -ge 20 ]]; then REC_TIER="32B" - elif [[ "$VRAM_GB" -ge 12 ]]; then REC_TIER="22B" - elif [[ "$VRAM_GB" -ge 6 ]]; then REC_TIER="14B" - else REC_TIER="7B" - fi - # performance pref has no 22B tier - [[ "$MODEL_PREF" == "2" && "$REC_TIER" == "22B" ]] && REC_TIER="32B" + case "$MODEL_PREF" in + 2) # Performance (Qwen 3.5) + if [[ "$TOTAL_VRAM" -ge 28 ]]; then REC_TIER="27B" + elif [[ "$TOTAL_VRAM" -ge 14 ]]; then REC_TIER="35B" + elif [[ "$TOTAL_VRAM" -ge 8 ]]; then REC_TIER="9B" + else REC_TIER="4B" + fi + ;; + *) # Western / Mixed + if [[ "$TOTAL_VRAM" -ge 40 ]]; then REC_TIER="70B" + elif [[ "$TOTAL_VRAM" -ge 12 ]]; then REC_TIER="35B" + elif [[ "$TOTAL_VRAM" -ge 8 ]]; then REC_TIER="14B" + else REC_TIER="7B" + fi + ;; + esac echo "" echo " Size tier β€” estimated speed on your ${VRAM_GB}GB GPU:" @@ -1391,6 +1403,14 @@ if $INSTALL_AI; then echo " Via MCP: gitea_github_sync(mode='all')" } fi +if $INSTALL_AI; then + echo "" + echo -e " ${YELLOW}Recommended Open WebUI Functions (install from Admin β†’ Functions β†’ οΌ‹):${NC}" + echo " Context tracker: https://openwebui.com/f/centrisic/context_tracker" + echo " β†’ Shows tokens used vs available, progress bar, context % remaining" + echo " Context compaction: https://openwebui.com/f/projectmoon/checkpoint_summarization_filter" + echo " β†’ Auto-summarizes old messages when context fills up (like Claude)" +fi if $SVC_KIWIX && [[ "$ZIM_CHOICE" == "3" ]]; then echo -e " ${YELLOW}ZIM downloads:${NC} ./kiwix_download.sh (not started)" elif $SVC_KIWIX; then diff --git a/local-ai-setup.sh b/local-ai-setup.sh index ceff246..675952c 100755 --- a/local-ai-setup.sh +++ b/local-ai-setup.sh @@ -42,6 +42,9 @@ elif [[ "$TOTAL_VRAM" -ge 14 ]]; then elif [[ "$TOTAL_VRAM" -ge 8 ]]; then CHAT_MODEL="qwen3.5:9b"; CODE_MODEL="qwen3.5:9b" CTX=32768; TIER="${TOTAL_VRAM}GB β€” 9B dense, 32K context" +elif [[ "$TOTAL_VRAM" -ge 6 ]]; then + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=32768; TIER="${TOTAL_VRAM}GB β€” 4B + KV quant, 32K context" elif [[ "$TOTAL_VRAM" -ge 4 ]]; then CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" CTX=16384; TIER="${TOTAL_VRAM}GB β€” 4B models, 16K context" @@ -898,7 +901,13 @@ echo "" echo -e " ${YELLOW}Aider CLI:${NC} cd your-repo && $BASE/aider.sh " echo " (or open browser UI above β€” both use your local code model)" echo "" -echo -e " ${YELLOW}Save Claude usage:${NC} use local 14B for boilerplate, docs," -echo " simple fixes. Use Claude Sonnet 4.6 for hard bugs," +echo -e " ${YELLOW}Recommended Open WebUI Functions${NC} (install from Admin β†’ Functions β†’ οΌ‹):" +echo " Context tracker: https://openwebui.com/f/centrisic/context_tracker" +echo " β†’ Shows tokens used vs available, progress bar, context % remaining" +echo " Context compaction: https://openwebui.com/f/projectmoon/checkpoint_summarization_filter" +echo " β†’ Auto-summarizes old messages when context fills up (like Claude)" +echo "" +echo -e " ${YELLOW}Save Claude usage:${NC} use local models for boilerplate, docs," +echo " simple fixes. Use Claude for hard bugs," echo " multi-file refactoring, architecture decisions." echo "" diff --git a/ubuntu-post-install.sh b/ubuntu-post-install.sh index eeda5ad..80ca43e 100644 --- a/ubuntu-post-install.sh +++ b/ubuntu-post-install.sh @@ -1125,6 +1125,9 @@ elif [[ "$TOTAL_VRAM" -ge 14 ]]; then elif [[ "$TOTAL_VRAM" -ge 8 ]]; then CHAT_MODEL="qwen3.5:9b"; CODE_MODEL="qwen3.5:9b" CTX=32768; TIER="${TOTAL_VRAM}GB β€” 9B dense, 32K context" +elif [[ "$TOTAL_VRAM" -ge 6 ]]; then + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=32768; TIER="${TOTAL_VRAM}GB β€” 4B + KV quant, 32K context" elif [[ "$TOTAL_VRAM" -ge 4 ]]; then CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" CTX=16384; TIER="${TOTAL_VRAM}GB β€” 4B models, 16K context"