[{"data":1,"prerenderedAt":1206},["ShallowReactive",2],{"navigation_docs":3,"-machine-learning-decision-tree-complete-guide":352,"-machine-learning-decision-tree-complete-guide-surround":1201},[4,54,83,116,125,161,179,200],{"title":5,"path":6,"stem":7,"children":8},"Ai","/ai","ai",[9,12,42,46,50],{"title":10,"path":6,"stem":11},"AI 技術探索","ai/index",{"title":13,"path":14,"stem":15,"children":16,"page":41},"Rag","/ai/rag","ai/RAG",[17,21,25,29,33,37],{"title":18,"path":19,"stem":20},"BM25 演算法深入解析：從 TF-IDF 到現代搜尋引擎的核心技術","/ai/rag/bm25-deep-dive-from-tf-idf-to-rag","ai/RAG/bm25-deep-dive-from-tf-idf-to-rag",{"title":22,"path":23,"stem":24},"Dense Retrieval（向量檢索）深入解析","/ai/rag/dense-retrieval-deep-dive","ai/RAG/dense-retrieval-deep-dive",{"title":26,"path":27,"stem":28},"RAG（Retrieval-Augmented Generation）原理與實作入門","/ai/rag/rag-intro-principles-and-implementation","ai/RAG/rag-intro-principles-and-implementation",{"title":30,"path":31,"stem":32},"RAG Retrieval Strategy 深入解析","/ai/rag/rag-retrieval-strategy-deep-dive","ai/RAG/rag-retrieval-strategy-deep-dive",{"title":34,"path":35,"stem":36},"Reciprocal Rank Fusion（RRF）教學：打造更穩定的檢索融合策略","/ai/rag/rrf-reciprocal-rank-fusion-guide","ai/RAG/rrf-reciprocal-rank-fusion-guide",{"title":38,"path":39,"stem":40},"TF-IDF（Term Frequency–Inverse Document Frequency）深入解析","/ai/rag/tf-idf-deep-dive","ai/RAG/tf-idf-deep-dive",false,{"title":43,"path":44,"stem":45},"Docus AI 功能完整實作指南","/ai/docus-ai-implementation","ai/docus-ai-implementation",{"title":47,"path":48,"stem":49},"llms.txt 是什麼？在 Nuxt 中控制 AI 與搜尋引擎爬蟲的新方式","/ai/llms-txt","ai/llms-txt",{"title":51,"path":52,"stem":53},"MCP（Model Context Protocol）是什麼？用「AI 的 USB-C」理解模型如何安全連接工具與資料","/ai/mcp_concept","ai/mcp_concept",{"title":55,"path":56,"stem":57,"children":58,"page":41},"Data Analysis","/data-analysis","data-analysis",[59,63,67,71,75,79],{"title":60,"path":61,"stem":62},"Exploratory Data Analysis（EDA）資料探索分析入門","/data-analysis/exploratory-data-analysis-eda","data-analysis/exploratory-data-analysis-eda",{"title":64,"path":65,"stem":66},"PR-AUC 深入解析：不平衡分類問題的重要評估指標","/data-analysis/pr-auc-for-imbalanced-classification","data-analysis/pr-auc-for-imbalanced-classification",{"title":68,"path":69,"stem":70},"ROC-AUC 深入解析：理解分類模型的判斷能力","/data-analysis/roc-auc-complete-guide","data-analysis/roc-auc-complete-guide",{"title":72,"path":73,"stem":74},"SHAP（SHapley Additive exPlanations）模型解釋方法深入解析","/data-analysis/shap-model-interpretation-complete-guide","data-analysis/shap-model-interpretation-complete-guide",{"title":76,"path":77,"stem":78},"Social Network Analysis（SNA）深入解析","/data-analysis/social-network-analysis-sna","data-analysis/social-network-analysis-sna",{"title":80,"path":81,"stem":82},"Time-based Split：為什麼時間序列資料不能使用 Random Split？","/data-analysis/time-based-split-vs-random-split","data-analysis/time-based-split-vs-random-split",{"title":84,"path":85,"stem":86,"children":87,"page":41},"MachineLearning","/machine_learning","machine_learning",[88,92,96,100,104,108,112],{"title":89,"path":90,"stem":91},"Confusion Matrix（混淆矩陣）完整解析","/machine_learning/confusion-matrix","machine_learning/confusion-matrix",{"title":93,"path":94,"stem":95},"Decision Tree 模型完整教學：從原理到 Python 實作","/machine_learning/decision-tree-complete-guide","machine_learning/decision-tree-complete-guide",{"title":97,"path":98,"stem":99},"為什麼 Decision Tree 在 Tabular Data 上常常勝過 Deep Learning？","/machine_learning/decision-tree-vs-deep-learning-on-tabular-data","machine_learning/decision-tree-vs-deep-learning-on-tabular-data",{"title":101,"path":102,"stem":103},"LightGBM 模型入門：理解高效能的 Gradient Boosting 演算法","/machine_learning/lightgbm-intro-for-tabular-data","machine_learning/lightgbm-intro-for-tabular-data",{"title":105,"path":106,"stem":107},"Logistic Regression（Logit 回歸）完整教學","/machine_learning/logistic-regression-complete-guide","machine_learning/logistic-regression-complete-guide",{"title":109,"path":110,"stem":111},"為什麼模型訓練需要 Train / Validation / Test Dataset？","/machine_learning/train-validation-test-dataset","machine_learning/train-validation-test-dataset",{"title":113,"path":114,"stem":115},"XGBoost 原理與實務教學","/machine_learning/xgboost-complete-guide","machine_learning/xgboost-complete-guide",{"title":117,"path":118,"stem":119,"children":120,"page":41},"Network","/network","network",[121],{"title":122,"path":123,"stem":124},"DNS 設定入門：CNAME vs A Record 完整解析","/network/dns-cname-vs-a-record-guide","network/dns-cname-vs-a-record-guide",{"title":126,"path":127,"stem":128,"children":129},"Nuxt","/nuxt","nuxt",[130,133,137,141,145,149,153,157],{"title":131,"path":127,"stem":132},"Nuxt 實戰指南","nuxt/index",{"title":134,"path":135,"stem":136},"Abstraction Layer 與 Adapter 的差別，一次搞懂設計角色","/nuxt/adapter_abstraction_layer","nuxt/adapter_abstraction_layer",{"title":138,"path":139,"stem":140},"Nuxt 4 全面理解 — 專案檔案結構解析","/nuxt/directory_structure","nuxt/directory_structure",{"title":142,"path":143,"stem":144},"Nuxt 前端 build 流程完整解析","/nuxt/frontend_build","nuxt/frontend_build",{"title":146,"path":147,"stem":148},"Nitro 是什麼？從 Nuxt 專案到 Server 與 Edge 的關鍵引擎","/nuxt/nitro","nuxt/nitro",{"title":150,"path":151,"stem":152},"深入理解 Nuxt 的 .nuxt 與 .output：為什麼部署時一定要分清楚？","/nuxt/nuxt_and_output","nuxt/nuxt_and_output",{"title":154,"path":155,"stem":156},"Nuxt 中 dev、build、preview 的差異一次搞懂","/nuxt/nuxt_dev_build_preview","nuxt/nuxt_dev_build_preview",{"title":158,"path":159,"stem":160},"Nuxt Image vs \u003Cimg>：為什麼 Nuxt 3 專案幾乎都該用 \u003CNuxtImg>？","/nuxt/nuxt_image","nuxt/nuxt_image",{"title":162,"path":163,"stem":164,"children":165},"Python 程式開發","/python","python/index",[166,167,171,175],{"title":162,"path":163,"stem":164},{"title":168,"path":169,"stem":170},"如何在 FastAPI 中調用記憶體 Buffer 回傳圖片 (附實例)","/python/memory_buffer","python/memory_buffer",{"title":172,"path":173,"stem":174},"Python 單元測試教學：如何為你的程式撰寫 Test","/python/python-unit-testing-unittest-pytest","python/python-unit-testing-unittest-pytest",{"title":176,"path":177,"stem":178},"如何使用 uv 取代 pip：改善 Python 專案的開發流程","/python/uv","python/uv",{"title":180,"path":181,"stem":182,"children":183,"page":41},"Statistic","/statistic","statistic",[184,188,192,196],{"title":185,"path":186,"stem":187},"卡方檢定（Chi-Square Test）入門教學","/statistic/chi-square-test-intro","statistic/chi-square-test-intro",{"title":189,"path":190,"stem":191},"常見抽樣方法（Sampling Methods）教學：Stratified、Quota、Convenience、Systematic、Simple Random","/statistic/common-sampling-methods-guide","statistic/common-sampling-methods-guide",{"title":193,"path":194,"stem":195},"Cramér’s V 指標介紹：如何衡量兩個類別變數之間的關聯","/statistic/cramers-v-for-categorical-association","statistic/cramers-v-for-categorical-association",{"title":197,"path":198,"stem":199},"Mann–Whitney U 與 Kolmogorov–Smirnov（KS）檢定：非參數統計檢定的入門指南","/statistic/mann-whitney-u-and-ks-test-intro","statistic/mann-whitney-u-and-ks-test-intro",{"title":201,"path":202,"stem":203,"children":204},"WebDev","/web_dev","web_dev",[205,208,234,248,282,312,334],{"title":206,"path":202,"stem":207},"Web 前端開發","web_dev/index",{"title":209,"path":210,"stem":211,"children":212},"瀏覽器與渲染","/web_dev/browser","web_dev/browser/index",[213,214,218,222,226,230],{"title":209,"path":210,"stem":211},{"title":215,"path":216,"stem":217},"瀏覽器儲存空間完整解析：Cookie、localStorage、IndexedDB 到 Cache Storage","/web_dev/browser/browser-storage-comprehensive-guide","web_dev/browser/browser-storage-comprehensive-guide",{"title":219,"path":220,"stem":221},"DOM (Document Object Model) 深入解析","/web_dev/browser/dom","web_dev/browser/dom",{"title":223,"path":224,"stem":225},"Service Worker Request Flow 深入解析","/web_dev/browser/service-worker-request-flow","web_dev/browser/service-worker-request-flow",{"title":227,"path":228,"stem":229},"CSR、SSR 與 SSG 是什麼？前端渲染策略完整比較","/web_dev/browser/ssr_csr_ssg","web_dev/browser/ssr_csr_ssg",{"title":231,"path":232,"stem":233},"Virtual DOM 深入解析：為什麼它能優化前端效能？","/web_dev/browser/virtual_dom","web_dev/browser/virtual_dom",{"title":235,"path":236,"stem":237,"children":238},"圖形技術","/web_dev/graphics","web_dev/graphics/index",[239,240,244],{"title":235,"path":236,"stem":237},{"title":241,"path":242,"stem":243},"為什麼 Three.js 專案幾乎都選擇 CSR？從 SSR 問題談起","/web_dev/graphics/threejs_csr","web_dev/graphics/threejs_csr",{"title":245,"path":246,"stem":247},"WebGL 是什麼？為什麼前端 3D 幾乎都靠它？","/web_dev/graphics/webgl","web_dev/graphics/webGL",{"title":249,"path":250,"stem":251,"children":252},"架構與配置","/web_dev/infrastructure","web_dev/infrastructure/index",[253,254,258,262,266,270,274,278],{"title":249,"path":250,"stem":251},{"title":255,"path":256,"stem":257},"Zeabur + K3s + Nuxt 部署架構完整解析","/web_dev/infrastructure/k3s-zeabur","web_dev/infrastructure/K3s-zeabur",{"title":259,"path":260,"stem":261},"ECS vs Docker vs Kubernetes：從部署堆疊理解三層架構","/web_dev/infrastructure/ecs-docker-kubernetes-stack","web_dev/infrastructure/ecs-docker-kubernetes-stack",{"title":263,"path":264,"stem":265},"ECS vs VPS 是什麼？從部署網站的角度一次搞懂差別","/web_dev/infrastructure/ecs-vs-vps","web_dev/infrastructure/ecs-vs-vps",{"title":267,"path":268,"stem":269},"Zeabur 內網服務連線完整入門指南","/web_dev/infrastructure/k3s-internal-networking","web_dev/infrastructure/k3s-internal-networking",{"title":271,"path":272,"stem":273},"使用 Microservices 架構設計系統的優勢解析","/web_dev/infrastructure/microservices-architecture-advantages","web_dev/infrastructure/microservices-architecture-advantages",{"title":275,"path":276,"stem":277},"Nginx 入門教學：從反向代理到與 Kubernetes 的架構比較","/web_dev/infrastructure/nginx-intro","web_dev/infrastructure/nginx-intro",{"title":279,"path":280,"stem":281},"YAML 配置是什麼？為什麼現代開發都在用它","/web_dev/infrastructure/yaml-configuration","web_dev/infrastructure/yaml-configuration",{"title":283,"path":284,"stem":285,"children":286},"網絡與通訊","/web_dev/network","web_dev/network/index",[287,288,292,296,300,304,308],{"title":283,"path":284,"stem":285},{"title":289,"path":290,"stem":291},"HTTP Request 結構完整解析：從 Request Line 到 Header 一次看懂","/web_dev/network/http-request-structure","web_dev/network/http-request-structure",{"title":293,"path":294,"stem":295},"OSI 模型（Open Systems Interconnection Model）完整解析","/web_dev/network/osi-model","web_dev/network/osi-model",{"title":297,"path":298,"stem":299},"RESTful API 設計原則與實務解析","/web_dev/network/restful-api","web_dev/network/restful-api",{"title":301,"path":302,"stem":303},"RESTful API 五大設計原則深入解析","/web_dev/network/restful-api-principles","web_dev/network/restful-api-principles",{"title":305,"path":306,"stem":307},"Server-Sent Events (SSE) 深入解析與實作教學","/web_dev/network/sse-introduction","web_dev/network/sse-introduction",{"title":309,"path":310,"stem":311},"WebSocket 入門教學：從概念到 Node.js 實作","/web_dev/network/websocket-introduction","web_dev/network/websocket-introduction",{"title":313,"path":314,"stem":315,"children":316},"認證與安全","/web_dev/security","web_dev/security/index",[317,318,322,326,330],{"title":313,"path":314,"stem":315},{"title":319,"path":320,"stem":321},"使用 Cookie 與 Session 建立使用者驗證（完整新手教學）","/web_dev/security/cookie-session-authentication","web_dev/security/cookie-session-authentication",{"title":323,"path":324,"stem":325},"JWT 驗證機制完整解析：從登入流程到實務應用","/web_dev/security/jwt-authentication","web_dev/security/jwt-authentication",{"title":327,"path":328,"stem":329},"Secret Key 簽名是什麼？從零理解資料簽名的本質","/web_dev/security/secret-key-signing","web_dev/security/secret-key-signing",{"title":331,"path":332,"stem":333},"SSH（Secure Shell）是什麼？從遠端登入到安全通道的核心概念","/web_dev/security/ssh","web_dev/security/ssh",{"title":335,"path":336,"stem":337,"children":338},"SEO 與規範","/web_dev/seo","web_dev/seo/index",[339,340,344,348],{"title":335,"path":336,"stem":337},{"title":341,"path":342,"stem":343},"Canonical URL 是什麼？用生活化方式搞懂前端 SEO 的基本保命符","/web_dev/seo/canonical_link","web_dev/seo/canonical_link",{"title":345,"path":346,"stem":347},"robots.txt 是什麼？SEO 的第一道守門員","/web_dev/seo/robot_txt","web_dev/seo/robot_txt",{"title":349,"path":350,"stem":351},"Nuxt SEO 中的 Meta 與 SEO 工具是如何運作的？","/web_dev/seo/seo","web_dev/seo/seo",{"id":353,"title":93,"body":354,"description":1190,"extension":1191,"links":1192,"meta":1193,"navigation":766,"path":94,"seo":1199,"stem":95,"__hash__":1200},"docs/machine_learning/decision-tree-complete-guide.md",{"type":355,"value":356,"toc":1164},"minimark",[357,362,371,374,382,392,395,408,415,418,422,425,430,433,439,442,446,449,455,458,462,465,471,474,476,480,483,569,572,578,581,593,596,598,602,605,612,622,625,628,631,634,641,645,648,651,654,661,663,667,670,676,679,690,693,695,699,706,709,894,897,901,906,981,984,990,993,995,999,1002,1022,1025,1031,1034,1036,1040,1043,1047,1050,1056,1059,1073,1077,1080,1082,1086,1089,1092,1106,1116,1118,1121,1124,1127,1133,1136,1138,1141,1160],[358,359,361],"h2",{"id":360},"decision-tree-是什麼","Decision Tree 是什麼？",[363,364,365,366,370],"p",{},"Decision Tree（決策樹）是一種常見的 ",[367,368,369],"strong",{},"Tree-based model（樹模型）","，廣泛應用於分類（classification）與回歸（regression）問題中。它的核心概念非常直覺：模型會透過一連串的條件判斷，把資料逐步切分，最後形成一棵樹狀結構。",[363,372,373],{},"在模型訓練的過程中，Decision Tree 會不斷進行以下動作：首先選擇一個最適合的特徵（feature）作為分裂條件；接著依照這個條件將資料分成不同群組；然後在每個群組中重複同樣的過程。隨著不斷分裂，最終會形成一棵完整的決策樹。",[363,375,376,377,381],{},"這棵樹的邏輯其實非常接近我們日常寫程式時使用的 ",[378,379,380],"code",{},"if-else"," 判斷。例如，一個簡單的決策流程可能像這樣：",[383,384,390],"pre",{"className":385,"code":387,"language":388,"meta":389},[386],"language-text","Age \u003C 30?\n├─ Yes → Income \u003C 50k?\n│  ├─ Yes → Buy\n│  └─ No  → No Buy\n└─ No  → Buy\n","text","",[378,391,387],{"__ignoreMap":389},[363,393,394],{},"這個決策過程可以理解為：",[396,397,398,402,405],"ol",{},[399,400,401],"li",{},"先判斷「年齡是否小於 30」",[399,403,404],{},"如果是，再檢查收入",[399,406,407],{},"如果不是，則直接做出預測",[363,409,410,411,414],{},"因此，Decision Tree 本質上就是一組自動學習出來的 ",[367,412,413],{},"if-else 規則集合","。",[416,417],"hr",{},[358,419,421],{"id":420},"decision-tree-的基本結構","Decision Tree 的基本結構",[363,423,424],{},"一棵 Decision Tree 通常由三種類型的節點組成：根節點（Root Node）、內部節點（Internal Node）以及葉節點（Leaf Node）。理解這三種節點的角色，可以幫助我們更清楚掌握整個模型的運作方式。",[426,427,429],"h3",{"id":428},"root-node根節點","Root Node（根節點）",[363,431,432],{},"根節點是整棵樹的起點，也是模型做出的第一個分裂決策。這個節點會選擇一個最具區分能力的特徵來分割資料。例如：",[383,434,437],{"className":435,"code":436,"language":388,"meta":389},[386],"Age \u003C 30\n",[378,438,436],{"__ignoreMap":389},[363,440,441],{},"這個條件會把資料分成兩群：符合條件與不符合條件。",[426,443,445],{"id":444},"internal-node內部節點","Internal Node（內部節點）",[363,447,448],{},"內部節點位於樹的中間，負責進一步細分資料。每個節點都會包含一個判斷條件，例如：",[383,450,453],{"className":451,"code":452,"language":388,"meta":389},[386],"Income \u003C 50000\n",[378,454,452],{"__ignoreMap":389},[363,456,457],{},"透過這樣的條件判斷，資料會被持續分割成更小、更純的群組。",[426,459,461],{"id":460},"leaf-node葉節點","Leaf Node（葉節點）",[363,463,464],{},"葉節點是樹的終點，也就是模型的最終預測結果。例如在分類問題中，葉節點可能代表：",[383,466,469],{"className":467,"code":468,"language":388,"meta":389},[386],"Fraud\nNot Fraud\n",[378,470,468],{"__ignoreMap":389},[363,472,473],{},"當一筆資料走到某個葉節點時，模型就會輸出該節點所代表的預測結果。",[416,475],{},[358,477,479],{"id":478},"實際案例信用卡詐欺偵測","實際案例：信用卡詐欺偵測",[363,481,482],{},"為了更直觀地理解 Decision Tree 的運作方式，我們可以看一個簡化的金融詐欺偵測案例。假設銀行希望判斷每一筆交易是否為詐欺交易，資料可能包含以下欄位：",[484,485,486,505],"table",{},[487,488,489],"thead",{},[490,491,492,496,499,502],"tr",{},[493,494,495],"th",{},"Amount",[493,497,498],{},"Country",[493,500,501],{},"Night Transaction",[493,503,504],{},"Fraud",[506,507,508,522,534,546,557],"tbody",{},[490,509,510,514,517,520],{},[511,512,513],"td",{},"5000",[511,515,516],{},"US",[511,518,519],{},"Yes",[511,521,519],{},[490,523,524,527,529,532],{},[511,525,526],{},"20",[511,528,516],{},[511,530,531],{},"No",[511,533,531],{},[490,535,536,539,542,544],{},[511,537,538],{},"3000",[511,540,541],{},"China",[511,543,519],{},[511,545,519],{},[490,547,548,551,553,555],{},[511,549,550],{},"15",[511,552,516],{},[511,554,531],{},[511,556,531],{},[490,558,559,562,565,567],{},[511,560,561],{},"2000",[511,563,564],{},"Russia",[511,566,519],{},[511,568,519],{},[363,570,571],{},"在訓練過程中，Decision Tree 可能學到以下的決策結構：",[383,573,576],{"className":574,"code":575,"language":388,"meta":389},[386],"Transaction Amount > 1000 ?\n├─ Yes → Night transaction?\n│  ├─ Yes → Fraud\n│  └─ No  → Not Fraud\n└─ No  → Not Fraud\n",[378,577,575],{"__ignoreMap":389},[363,579,580],{},"這棵樹代表的邏輯是：",[582,583,584,587,590],"ul",{},[399,585,586],{},"如果交易金額小於 1000，模型通常會判斷為正常交易",[399,588,589],{},"如果交易金額大於 1000，模型會進一步檢查是否為夜間交易",[399,591,592],{},"若同時滿足高金額與夜間交易，則該交易很可能是詐欺",[363,594,595],{},"透過這樣的層層判斷，Decision Tree 可以逐步將不同類型的交易區分開來。",[416,597],{},[358,599,601],{"id":600},"decision-tree-如何決定分裂方式","Decision Tree 如何決定分裂方式？",[363,603,604],{},"一個重要問題是：模型如何知道應該先用哪個特徵來分裂資料？例如為什麼先使用交易金額，而不是交易國家？",[363,606,607,608,611],{},"Decision Tree 的答案是透過 ",[367,609,610],{},"資料純度（purity）指標"," 來決定最佳分裂方式。模型會嘗試不同的切分方法，並選擇能讓資料「最純」的那個。",[363,613,614,615,618,619,414],{},"常見的純度指標包含 ",[367,616,617],{},"Gini Impurity"," 與 ",[367,620,621],{},"Entropy",[426,623,617],{"id":624},"gini-impurity",[363,626,627],{},"Gini Impurity 是最常見的決策樹分裂指標，其公式為：",[363,629,630],{},"$$\nGini = 1 - \\sum p_i^2\n$$",[363,632,633],{},"其中 $p_i$ 代表某個類別在節點中的比例。如果一個節點中的資料幾乎全部屬於同一個類別，那麼 Gini 值就會非常小，代表節點非常純。",[363,635,636,637,640],{},"在訓練過程中，Decision Tree 會選擇能 ",[367,638,639],{},"最大程度降低 Gini 值"," 的分裂方式。",[426,642,644],{"id":643},"entropy-與-information-gain","Entropy 與 Information Gain",[363,646,647],{},"另一種常見方法是 Entropy，其公式為：",[363,649,650],{},"$$\nEntropy = -\\sum p_i \\log(p_i)\n$$",[363,652,653],{},"Entropy 代表資料的不確定性。當資料越混亂時，Entropy 越高；當資料越純時，Entropy 越低。",[363,655,656,657,660],{},"Decision Tree 會計算分裂前後的 Entropy 差異，這個差異稱為 ",[367,658,659],{},"Information Gain（資訊增益）","。模型會選擇 Information Gain 最大的分裂方式。",[416,662],{},[358,664,666],{"id":665},"decision-tree-的訓練流程","Decision Tree 的訓練流程",[363,668,669],{},"Decision Tree 的訓練過程可以簡化為以下步驟：",[383,671,674],{"className":672,"code":673,"language":388,"meta":389},[386],"Step 1: 選擇最佳 feature\nStep 2: 用這個 feature 分裂資料\nStep 3: 對每個子節點重複\nStep 4: 直到滿足停止條件\n",[378,675,673],{"__ignoreMap":389},[363,677,678],{},"模型會持續分裂資料，直到達到某些停止條件，例如：",[582,680,681,684,687],{},[399,682,683],{},"樹的深度已達上限",[399,685,686],{},"節點樣本數過少",[399,688,689],{},"節點中的資料已完全純化",[363,691,692],{},"當停止條件成立時，該節點就會成為葉節點。",[416,694],{},[358,696,698],{"id":697},"python-實際範例","Python 實際範例",[363,700,701,702,705],{},"在 Python 中，可以透過 ",[378,703,704],{},"scikit-learn"," 快速建立 Decision Tree 模型。以下示範使用經典的 Iris 資料集。",[426,707,708],{"id":708},"建立模型",[383,710,714],{"className":711,"code":712,"language":713,"meta":389,"style":389},"language-python shiki shiki-themes github-dark","from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_iris\n\ndata = load_iris()\nX = data.data\ny = data.target\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42\n)\n\nmodel = DecisionTreeClassifier(max_depth=3)\nmodel.fit(X_train, y_train)\npred = model.predict(X_test)\n","python",[378,715,716,735,748,761,768,780,791,802,807,818,845,851,856,877,883],{"__ignoreMap":389},[717,718,721,725,729,732],"span",{"class":719,"line":720},"line",1,[717,722,724],{"class":723},"snl16","from",[717,726,728],{"class":727},"s95oV"," sklearn.tree ",[717,730,731],{"class":723},"import",[717,733,734],{"class":727}," DecisionTreeClassifier\n",[717,736,738,740,743,745],{"class":719,"line":737},2,[717,739,724],{"class":723},[717,741,742],{"class":727}," sklearn.model_selection ",[717,744,731],{"class":723},[717,746,747],{"class":727}," train_test_split\n",[717,749,751,753,756,758],{"class":719,"line":750},3,[717,752,724],{"class":723},[717,754,755],{"class":727}," sklearn.datasets ",[717,757,731],{"class":723},[717,759,760],{"class":727}," load_iris\n",[717,762,764],{"class":719,"line":763},4,[717,765,767],{"emptyLinePlaceholder":766},true,"\n",[717,769,771,774,777],{"class":719,"line":770},5,[717,772,773],{"class":727},"data ",[717,775,776],{"class":723},"=",[717,778,779],{"class":727}," load_iris()\n",[717,781,783,786,788],{"class":719,"line":782},6,[717,784,785],{"class":727},"X ",[717,787,776],{"class":723},[717,789,790],{"class":727}," data.data\n",[717,792,794,797,799],{"class":719,"line":793},7,[717,795,796],{"class":727},"y ",[717,798,776],{"class":723},[717,800,801],{"class":727}," data.target\n",[717,803,805],{"class":719,"line":804},8,[717,806,767],{"emptyLinePlaceholder":766},[717,808,810,813,815],{"class":719,"line":809},9,[717,811,812],{"class":727},"X_train, X_test, y_train, y_test ",[717,814,776],{"class":723},[717,816,817],{"class":727}," train_test_split(\n",[717,819,821,824,828,830,834,837,840,842],{"class":719,"line":820},10,[717,822,823],{"class":727},"    X, y, ",[717,825,827],{"class":826},"s9osk","test_size",[717,829,776],{"class":723},[717,831,833],{"class":832},"sDLfK","0.2",[717,835,836],{"class":727},", ",[717,838,839],{"class":826},"random_state",[717,841,776],{"class":723},[717,843,844],{"class":832},"42\n",[717,846,848],{"class":719,"line":847},11,[717,849,850],{"class":727},")\n",[717,852,854],{"class":719,"line":853},12,[717,855,767],{"emptyLinePlaceholder":766},[717,857,859,862,864,867,870,872,875],{"class":719,"line":858},13,[717,860,861],{"class":727},"model ",[717,863,776],{"class":723},[717,865,866],{"class":727}," DecisionTreeClassifier(",[717,868,869],{"class":826},"max_depth",[717,871,776],{"class":723},[717,873,874],{"class":832},"3",[717,876,850],{"class":727},[717,878,880],{"class":719,"line":879},14,[717,881,882],{"class":727},"model.fit(X_train, y_train)\n",[717,884,886,889,891],{"class":719,"line":885},15,[717,887,888],{"class":727},"pred ",[717,890,776],{"class":723},[717,892,893],{"class":727}," model.predict(X_test)\n",[363,895,896],{},"在這個例子中，我們建立了一棵最大深度為 3 的決策樹模型，並使用訓練資料進行訓練。",[426,898,900],{"id":899},"視覺化-decision-tree","視覺化 Decision Tree",[363,902,903,905],{},[378,904,704],{}," 也提供了視覺化工具，可以將訓練好的樹結構畫出來。",[383,907,909],{"className":711,"code":908,"language":713,"meta":389,"style":389},"from sklearn.tree import plot_tree\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 6))\nplot_tree(model, feature_names=data.feature_names)\nplt.show()\n",[378,910,911,922,935,939,963,976],{"__ignoreMap":389},[717,912,913,915,917,919],{"class":719,"line":720},[717,914,724],{"class":723},[717,916,728],{"class":727},[717,918,731],{"class":723},[717,920,921],{"class":727}," plot_tree\n",[717,923,924,926,929,932],{"class":719,"line":737},[717,925,731],{"class":723},[717,927,928],{"class":727}," matplotlib.pyplot ",[717,930,931],{"class":723},"as",[717,933,934],{"class":727}," plt\n",[717,936,937],{"class":719,"line":750},[717,938,767],{"emptyLinePlaceholder":766},[717,940,941,944,947,949,952,955,957,960],{"class":719,"line":763},[717,942,943],{"class":727},"plt.figure(",[717,945,946],{"class":826},"figsize",[717,948,776],{"class":723},[717,950,951],{"class":727},"(",[717,953,954],{"class":832},"12",[717,956,836],{"class":727},[717,958,959],{"class":832},"6",[717,961,962],{"class":727},"))\n",[717,964,965,968,971,973],{"class":719,"line":770},[717,966,967],{"class":727},"plot_tree(model, ",[717,969,970],{"class":826},"feature_names",[717,972,776],{"class":723},[717,974,975],{"class":727},"data.feature_names)\n",[717,977,978],{"class":719,"line":782},[717,979,980],{"class":727},"plt.show()\n",[363,982,983],{},"畫出的決策樹可能會像這樣：",[383,985,988],{"className":986,"code":987,"language":388,"meta":389},[386],"petal length \u003C 2.5 ?\n├─ Yes → Setosa\n└─ No  → petal width \u003C 1.8 ?\n",[378,989,987],{"__ignoreMap":389},[363,991,992],{},"透過視覺化，我們可以清楚看到模型的決策流程。",[416,994],{},[358,996,998],{"id":997},"為什麼-tree-model-在-tabular-data-上表現很好","為什麼 Tree Model 在 Tabular Data 上表現很好？",[363,1000,1001],{},"在許多表格型資料（tabular data）的任務中，Tree-based model 往往能取得非常好的效果。原因包含以下幾點：",[582,1003,1004,1007,1013,1019],{},[399,1005,1006],{},"不需要進行 feature scaling",[399,1008,1009,1010],{},"可以自然處理 ",[367,1011,1012],{},"非線性關係（non-linear relationships）",[399,1014,1015,1016],{},"能自動學習 ",[367,1017,1018],{},"feature interaction",[399,1020,1021],{},"對類別變數相對友善",[363,1023,1024],{},"例如以下規則：",[383,1026,1029],{"className":1027,"code":1028,"language":388,"meta":389},[386],"if income > 50000 AND age \u003C 30\n",[378,1030,1028],{"__ignoreMap":389},[363,1032,1033],{},"這樣的條件在真實世界中很常見，但線性模型通常較難直接捕捉。",[416,1035],{},[358,1037,1039],{"id":1038},"decision-tree-的缺點","Decision Tree 的缺點",[363,1041,1042],{},"雖然 Decision Tree 非常直覺且容易理解，但它也有一些限制。",[426,1044,1046],{"id":1045},"_1-容易-overfitting","1) 容易 overfitting",[363,1048,1049],{},"如果不限制樹的成長，模型可能會不斷分裂資料，甚至記住每一筆訓練資料。例如：",[383,1051,1054],{"className":1052,"code":1053,"language":388,"meta":389},[386],"Age \u003C 30\nAge \u003C 29\nAge \u003C 28\n",[378,1055,1053],{"__ignoreMap":389},[363,1057,1058],{},"這會導致模型在訓練資料上表現很好，但在新資料上表現很差。因此通常會透過以下方法控制模型複雜度：",[582,1060,1061,1065,1070],{},[399,1062,1063],{},[378,1064,869],{},[399,1066,1067],{},[378,1068,1069],{},"min_samples_leaf",[399,1071,1072],{},"pruning（剪枝）",[426,1074,1076],{"id":1075},"_2-模型不穩定high-variance","2) 模型不穩定（high variance）",[363,1078,1079],{},"當訓練資料稍微改變時，整棵樹的結構可能完全不同。",[416,1081],{},[358,1083,1085],{"id":1084},"tree-model-的進化版本","Tree Model 的進化版本",[363,1087,1088],{},"Decision Tree 是許多強大機器學習模型的基礎。為了解決單棵樹的缺點，研究者發展出多種改進方法。",[363,1090,1091],{},"最常見的包括：",[582,1093,1094,1100],{},[399,1095,1096,1099],{},[367,1097,1098],{},"Random Forest","：透過建立多棵 Decision Tree 並進行投票來提高穩定性",[399,1101,1102,1105],{},[367,1103,1104],{},"Gradient Boosting Decision Tree（GBDT）","：讓每一棵新樹專門修正前一棵樹的錯誤",[363,1107,1108,1109,618,1112,1115],{},"在實務應用中，GBDT 又進一步發展出多個高效版本，例如 ",[367,1110,1111],{},"XGBoost",[367,1113,1114],{},"LightGBM","，在許多資料科學競賽與任務中都表現非常出色。",[416,1117],{},[358,1119,1120],{"id":1120},"總結",[363,1122,1123],{},"Decision Tree 是一種非常直觀且易於解釋的機器學習模型。它透過不斷選擇最佳特徵來分裂資料，最終形成一棵由多個條件判斷組成的決策樹。每條從根節點到葉節點的路徑，其實都代表了一組清楚的 if-else 規則。",[363,1125,1126],{},"簡單來說，Decision Tree 的核心概念可以用一句話概括：",[1128,1129,1130],"blockquote",{},[363,1131,1132],{},"Decision Tree 本質上就是一個自動學習規則的系統，它會從資料中找出一連串最有效的 if-else 判斷，並用這些規則來進行預測。",[363,1134,1135],{},"理解 Decision Tree 的運作原理，不僅能幫助我們掌握基礎機器學習模型，也能為後續學習 Random Forest、XGBoost 與 LightGBM 等進階模型打下良好的基礎。",[416,1137],{},[358,1139,1140],{"id":1140},"參考資料",[396,1142,1143,1151,1154],{},[399,1144,1145,1146,1150],{},"Hastie, Tibshirani, Friedman. ",[1147,1148,1149],"em",{},"The Elements of Statistical Learning",".",[399,1152,1153],{},"scikit-learn Documentation – Decision Trees.",[399,1155,1156,1157,1150],{},"Bishop, C. ",[1147,1158,1159],{},"Pattern Recognition and Machine Learning",[1161,1162,1163],"style",{},"html pre.shiki code .snl16, html code.shiki .snl16{--shiki-default:#F97583}html pre.shiki code .s95oV, html code.shiki .s95oV{--shiki-default:#E1E4E8}html pre.shiki code .s9osk, html code.shiki .s9osk{--shiki-default:#FFAB70}html pre.shiki code .sDLfK, html code.shiki .sDLfK{--shiki-default:#79B8FF}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}",{"title":389,"searchDepth":737,"depth":737,"links":1165},[1166,1167,1172,1173,1177,1178,1182,1183,1187,1188,1189],{"id":360,"depth":737,"text":361},{"id":420,"depth":737,"text":421,"children":1168},[1169,1170,1171],{"id":428,"depth":750,"text":429},{"id":444,"depth":750,"text":445},{"id":460,"depth":750,"text":461},{"id":478,"depth":737,"text":479},{"id":600,"depth":737,"text":601,"children":1174},[1175,1176],{"id":624,"depth":750,"text":617},{"id":643,"depth":750,"text":644},{"id":665,"depth":737,"text":666},{"id":697,"depth":737,"text":698,"children":1179},[1180,1181],{"id":708,"depth":750,"text":708},{"id":899,"depth":750,"text":900},{"id":997,"depth":737,"text":998},{"id":1038,"depth":737,"text":1039,"children":1184},[1185,1186],{"id":1045,"depth":750,"text":1046},{"id":1075,"depth":750,"text":1076},{"id":1084,"depth":737,"text":1085},{"id":1120,"depth":737,"text":1120},{"id":1140,"depth":737,"text":1140},"深入了解 Decision Tree（決策樹）的運作原理，包含模型結構、純度指標、實際案例與 Python 實作，幫助你掌握最經典的機器學習模型之一。","md",null,{"tags":1194,"category":86,"date":1198},[86,1195,1196,1197],"decision_tree","classification","tree_model","2026-03-08",{"title":93,"description":1190},"OG7uec8DmLdnCcQI5Ky6ohggVxKU41EW0mmvlAL_IlM",[1202,1204],{"title":89,"path":90,"stem":91,"description":1203,"children":-1},"從直觀概念到實際應用，深入理解 Confusion Matrix 的結構、評估指標與在詐欺偵測中的重要性。",{"title":97,"path":98,"stem":99,"description":1205,"children":-1},"深入理解 Decision Tree 與 GBDT 在表格資料上的優勢，解析為何在多數資料科學實務中，tree-based model 往往比深度學習模型表現更好。",1776690841627]