{"templateId":"markdown","sharedDataIds":{"sidebar":"sidebar-@l10n/ja/sidebars.yaml"},"props":{"metadata":{"markdoc":{"tagList":[]},"type":"markdown"},"seo":{"title":"Next Best Action (NBA) Recommendation","description":"Predict the optimal next action, channel, send time, or offer for every customer, built on AI Signals, Treasure AI's ML platform.","siteUrl":"https://docs.treasure.ai","lang":"en-US","jsonLd":{"@context":"https://schema.org","@graph":[{"@type":"Organization","@id":"https://www.treasure.ai/","name":"Treasure AI","url":"https://www.treasure.ai/","logo":"https://www.treasure.ai/hubfs/assets/images/logos/primary-logo.svg"},{"@type":"WebSite","@id":"https://docs.treasure.ai/#website","name":"Treasure AI Documentation","url":"https://docs.treasure.ai/","inLanguage":["en","ja"],"publisher":{"@id":"https://www.treasure.ai/"}}]}},"dynamicMarkdocComponents":[],"compilationErrors":[],"ast":{"$$mdtype":"Tag","name":"article","attributes":{},"children":[{"$$mdtype":"Tag","name":"Heading","attributes":{"level":1,"id":"next-best-action-nba-recommendation","__idx":0},"children":["Next Best Action (NBA) Recommendation"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Learn from past user interactions to recommend the next best action, channel, send time, or offer, for every customer."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":2,"id":"overview-and-use-cases","__idx":1},"children":["Overview and Use Cases"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["The Next Best Action (NBA) AI Signals solution looks at how your users responded to past marketing actions and learns which action is most likely to work for each user next time. Instead of sending the same email, offer, or channel to everyone, NBA matches each user's context to the action they are most likely to engage with. The output plugs into Treasure AI Master Segments and Journey orchestration, so marketers and CRM teams can personalize at scale without writing policy rules by hand."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Under the hood, NBA is a ",{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["contextual bandit"]},", a model family built for explore/exploit decisions. It reads historical interaction logs, evaluates many candidate policies offline, and picks the one most likely to outperform random or rules-based targeting. You don't need to hand-label \"good\" or \"bad\" actions; the reward signal (a click, an open, a purchase) is all NBA needs to improve."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Common use cases"]}]},{"$$mdtype":"Tag","name":"ul","attributes":{},"children":[{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Next best channel:"]}," email, push, SMS, or paid media"]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Send time optimization:"]}," the time of day or day of week most likely to earn a response"]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Next best offer:"]}," the best coupon, deal, or product recommendation from a candidate set"]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Content personalization:"]}," subject line, hero banner, or creative variant per user"]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Feature input for downstream models:"]}," NBA recommendations as signals for CLTV, churn, or lifecycle models"]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Who benefits most:"]}," Lifecycle marketers, CRM managers, and personalization teams who already run A/B tests but want a learned, per-user policy rather than a single one-size-fits-all winner."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"how-it-fits-with-the-other-ai-signals","__idx":2},"children":["How It Fits with the Other AI Signals"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Question you're asking"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Question you're asking"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Use"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Use"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Who matters, based on past purchasing?"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"/ja/products/customer-data-platform/machine-learning/ai-signals/rfm-ai-signals"},"children":["RFM AI Signals"]},": descriptive segments, no modeling required"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":["How likely is this customer to do X?"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Propensity Scoring AI Signals: a calibrated probability per event"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":["How much will this customer be worth?"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"/ja/products/customer-data-platform/machine-learning/ai-signals/cltv-ai-signals"},"children":["CLTV AI Signals"]},": a dollar forecast and percentile rank"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Which product should I recommend next?"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"/ja/products/customer-data-platform/machine-learning/ai-signals/nbp-ai-signals"},"children":["NBP AI Signals"]},": a ranked list of products, services, or content"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Which action should I take for this customer?"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["NBA AI Signals (you are here)"]},": a learned per-user policy across candidate actions"]}]}]}]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"em","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["NBA or NBP?"]}," NBP ranks items from a catalog by affinity: what is this person most likely to want. NBA learns a policy over a small set of marketing decisions you control, based on what has actually worked: which channel, which send time, which offer. NBP works from purchase and interaction history. NBA needs logged action and outcome pairs. The two compose well: let NBP choose the product, then let NBA choose how and when to pitch it."]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":2,"id":"quick-start","__idx":3},"children":["Quick Start"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Let's get you started with an example workflow. NBA runs as three functions on the Treasure AI ML Batch API:"]},{"$$mdtype":"Tag","name":"ul","attributes":{},"children":[{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}]}," searches models, preprocessing, and evaluation settings to find the best configuration for your data."]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]}]}," fits the chosen policy on the full dataset and saves it to model storage."]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_predict"]}]}," scores users with a trained policy and writes the recommended action."]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Run tuning once to find a good configuration, then schedule training and prediction. Teams can tune monthly, or when they add new actions or change the feature set."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"tune","__idx":4},"children":["Tune"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"# Step 1: Tune, search for the best model and configuration\n+tune:\n  http>: https://ml-batch-api.treasuredata.com/v1/runs/\n  method: POST\n  headers:\n    - authorization: ${secret:td.apikey}\n    - X-TD-ML-SESSION-ID: ${session_id}\n    - X-TD-ML-ATTEMPT-ID: ${attempt_id}\n  store_content: true\n  content:\n    input_table: your_database.user_interactions\n    output_table: your_database.nba_tune_results\n    solution_name: nba_tune\n    solution_arguments:\n      action_column: \"item_id\"\n      reward_column: \"click\"\n      timestamp_column: \"timestamp\"\n      tune_ocv: true\n\n+tune_status:\n  http>: https://ml-batch-api.treasuredata.com/v1/runs/${JSON.parse(http.last_content)['id']}/status\n  method: GET\n  headers:\n    - authorization: ${secret:td.apikey}\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Expected output:"]}," a diagnostic table in ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["your_database.nba_tune_results"]},", one row per trial across three tuning phases, plus a random baseline."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"train","__idx":5},"children":["Train"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"# Step 2: Train, fit the selected policy on the full dataset\n+train:\n  http>: https://ml-batch-api.treasuredata.com/v1/runs/\n  method: POST\n  headers:\n    - authorization: ${secret:td.apikey}\n    - X-TD-ML-SESSION-ID: ${session_id}\n    - X-TD-ML-ATTEMPT-ID: ${attempt_id}\n  store_content: true\n  content:\n    input_table: your_database.user_interactions\n    output_table: your_database.nba_train_results\n    solution_name: nba_train\n    solution_arguments:\n      model_name: \"nba_retail_v1\"\n      action_column: \"item_id\"\n      reward_column: \"click\"\n      timestamp_column: \"timestamp\"\n      tuning_results_table: your_database.nba_tune_results\n\n+train_status:\n  http>: https://ml-batch-api.treasuredata.com/v1/runs/${JSON.parse(http.last_content)['id']}/status\n  method: GET\n  headers:\n    - authorization: ${secret:td.apikey}\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Expected output:"]}," a small metadata table confirming training completed. The trained policy is saved to managed model storage under ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_name"]}," and is retrieved by ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_predict"]}," automatically using the ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_name"]}," field. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_name"]}," must be unique per TD account."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["By default, ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_type"]}," comes from your tuning results. Passing ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_results_table"]}," reuses the tuned hyperparameters instead of falling back to generic defaults, which is the recommended path."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"predict","__idx":6},"children":["Predict"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"# Step 3: Predict, score users with the trained policy\n+predict:\n  http>: https://ml-batch-api.treasuredata.com/v1/runs/\n  method: POST\n  headers:\n    - authorization: ${secret:td.apikey}\n    - X-TD-ML-SESSION-ID: ${session_id}\n    - X-TD-ML-ATTEMPT-ID: ${attempt_id}\n  store_content: true\n  content:\n    input_table: your_database.users_to_score\n    output_table: your_database.nba_predictions\n    solution_name: nba_predict\n    solution_arguments:\n      model_name: \"nba_retail_v1\"\n      user_column: \"user_id\"\n      timestamp_column: \"timestamp\"\n\n+predict_status:\n  http>: https://ml-batch-api.treasuredata.com/v1/runs/${JSON.parse(http.last_content)['id']}/status\n  method: GET\n  headers:\n    - authorization: ${secret:td.apikey}\n\n+done:\n  echo>: \"NBA pipeline complete. Predictions in your_database.nba_predictions\"\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Expected output:"]}," one row per user with the recommended action."]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"header":{"controls":{"copy":{}}},"source":"time        user_id    predictions\n----------  ---------  -------------\n1758004803  23492371   [\"15\"]\n"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["To create and schedule these workflows, see ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"/products/customer-data-platform/data-workbench/workflows/getting-started-with-treasure-workflow"},"children":["Getting Started with Treasure Workflow"]},"."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":2,"id":"model-configuration","__idx":7},"children":["Model Configuration"]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"nba_tune","__idx":8},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," evaluates preprocessing, estimation, and policy settings to determine the best configuration for your data. Pass the tuning result to ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]}," via the ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_results_table"]}," parameter to automatically apply the optimal model type and hyperparameters."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"data-preparation","__idx":9},"children":["Data Preparation"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["The ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["input_table"]}," required for the tuning step is a ",{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["user-action interaction table"]},", structured with one row per event. This table should match the dataset you intend to use for the full training phase."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"em","attributes":{},"children":["The ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," model type is not part of the tuning search space. It is configured and trained directly through ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]},"."]}]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"15%","data-label":"Field"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Field"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"10%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"10%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"65%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["timestamp"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["LONG"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Unix timestamp of the interaction."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_id"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["VARCHAR"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Unique user identifier. Column name is customizable."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["action"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["VARCHAR"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["The action the user was shown, for example email, paid_search, a coupon code, or a numeric index."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["reward"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["INT"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Outcome of the interaction. Typically 1 for a click, open, or purchase, and 0 otherwise."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["feature_1 … feature_N"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["DOUBLE"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Columns describing the user's context: age, device, tenure, one-hot encoded profile fields, or a latent vector. Any number is supported."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["pscore"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["DOUBLE"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["The true propensity, meaning the probability the logging policy chose this action for this user. NBA estimates it if omitted."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["position"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["INT"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Position the action was shown in. Reserved for future multi-action support."]}]}]}]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Example rows:"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"header":{"controls":{"copy":{}}},"source":"timestamp   user_id    action       reward  age  tenure_days  device_mobile  pscore\n1757900000  23492371   email        1       34   412          1              0.25\n1757901200  88104553   push         0       51   87           0              0.25\n1757902450  23492371   sms          0       34   412          1              0.25\n1757903900  41029887   paid_search  1       27   1203         1              0.25\n"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Data requirements:"]}]},{"$$mdtype":"Tag","name":"ul","attributes":{},"children":[{"$$mdtype":"Tag","name":"li","attributes":{},"children":["Feature columns must be numeric. Encode categorical variables using one-hot, target encoding, or embeddings before passing them in."]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":["Aim for at least a few hundred interactions per action. NBA cannot reliably evaluate an action that appears only a handful of times."]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":["Positive rewards below roughly 1% of rows weaken both the propensity and reward models. More data and stronger features help."]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":["Missing values are handled by the configured imputer. See ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["impute_type"]}," in ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#nba_train"},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]}]},"."]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":["Use ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["exclude_columns"]}," to drop ID-like columns, raw timestamps, and leaky features."]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-parameters","__idx":10},"children":["Workflow Parameters"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"22%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["input_table"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Source table in dbname.table_name format."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["output_table"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Destination table for function output."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_id"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["user_id"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column holding the user identifier."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["action_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["action"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column holding the action taken."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["reward_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["reward"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column holding the reward."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["timestamp_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["timestamp"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Event time column."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["propensity_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column with true propensity scores. If omitted, NBA estimates them."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["exclude_columns"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Pipe-delimited patterns to drop from features, for example ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["col_a|*_raw|temp_*"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["hyperparam_tune_sample_ratio"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.01"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Fraction of data used for tuning. Increase for small datasets."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ocv_sigma_coef"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Conservativeness of policy selection. Higher values favor safer policies."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ocv_phase1_trials"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["20"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Optuna trials for OPE estimator tuning."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ocv_phase2_trials"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["50"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Optuna trials for policy tuning."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["search_space"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["object"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Override parts of the default hyperparameter search space."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ocv_ess_config"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["object"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Effective Sample Size filtering for OCV Phase 2. See ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#effective-sample-size-ess-configuration"},"children":["Effective Sample Size (ESS) Configuration"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["max_ope_samples"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["200000"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Maximum samples used for off-policy evaluation. Prevents out-of-memory errors on large datasets through stratified subsampling that guarantees action coverage."]}]}]}]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Keep ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tune_ocv: true"]},". OCV is paper-backed, gives more reliable policy selection, and is the recommended setting for new deployments."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"output","__idx":11},"children":["Output"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["A diagnostic table, one row per trial across three phases (preprocessing, OPE estimator selection, policy selection), plus a random baseline."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"nba_train","__idx":12},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]}," fits one chosen policy on the full dataset and saves it for reuse."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"data-preparation-1","__idx":13},"children":["Data Preparation"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["The ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["input_table"]}," required for the training step is a ",{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["user-action interaction table"]},", structured with one row per event. This is the same table you used for model tuning."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-parameters-1","__idx":14},"children":["Workflow Parameters"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"22%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["input_table"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Source table in dbname.table_name format."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["output_table"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Destination table for function output."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_type"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Policy family: lin_ucb, lin_ts, lin_eps_greedy, ipw_learner, or neural_lin_ucb. See ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#model-types"},"children":["Model Types"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_name"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Unique name used to save and retrieve the trained model. Unique per TD account."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_id"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["user_id"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column holding the user identifier."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["action_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["action"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column holding the action taken."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["reward_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["reward"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column holding the reward."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["timestamp_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["timestamp"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Event time column."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["propensity_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column with true propensity scores. If omitted, NBA estimates them."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["exclude_columns"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Pipe-delimited patterns to drop from features, for example ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["col_a|*_raw|temp_*"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["n_predictions"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Number of recommended actions per user. Currently fixed at 1; multi-prediction support is planned."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["epsilon"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Exploration rate for online models. Higher means more random exploration. Ignored by ipw_learner."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["impute_type"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["knn"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Missing-value strategy: knn, hybrid, median, most_frequent, mean, drop, none. knn is safe but slow on very large tables; try median if runtime matters."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["scaler_type"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["minmax"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Feature scaling: minmax, standard, robust, maxabs, none."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["base_classifier"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["random_forest"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Classifier inside ipw_learner: random_forest or logistic_regression."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["propensity_type"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["logistic"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Propensity estimation method: uniform, logistic, true_propensity."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tune_propensity"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["boolean"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["true"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Whether to tune propensity model hyperparameters."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["max_ope_samples"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["200000"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Maximum samples used for off-policy evaluation."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["max_training_samples_per_model"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["object"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["ipw_learner and neural_lin_ucb: 1000000; lin_ucb, lin_ts, lin_eps_greedy: 5000000"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Maximum training samples per model type. See ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#controlling-training-data-size"},"children":["Controlling Training Data Size"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["arm_feature_columns"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Conditional"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Pipe-separated glob patterns identifying static per-action attribute columns. Required when ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["per_arm=true"]},"; ignored with a warning by other model types."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb_params"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["object"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Hyperparameter block for neural_lin_ucb. See ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#neurallinucb--neural_lin_ucb-"},"children":["NeuralLinUCB"]},". Ignored by other model types."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_results_table"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Table of tuning results to source hyperparameters from, in database.table format. See ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#reuse-tuned-hyperparameters"},"children":["Reuse Tuned Hyperparameters"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_run_id"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Specific tuning run to use. Defaults to the most recent run in ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_results_table"]},"."]}]}]}]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"output-1","__idx":15},"children":["Output"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["A small metadata table confirming training completed. The trained policy itself is saved to a managed model storage under ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_name"]}," and retrieved by ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_predict"]}," automatically."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"nba_predict","__idx":16},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_predict"]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_predict"]}," loads a trained policy and scores users."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"data-preparation-2","__idx":17},"children":["Data Preparation"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["A table of user context vectors, one row per user. It carries the same feature columns used in training, but no ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["action"]}," and no ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["reward"]},"."]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"15%","data-label":"Field"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Field"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"10%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"10%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"65%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_id"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["VARCHAR"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Unique user identifier. Column name is customizable via ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_column"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["feature_1 … feature_N"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["DOUBLE"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["The same context feature columns used at training time."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["pscore"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["DOUBLE"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["True propensity score of the action proposed by the production policy, if known."]}]}]}]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Example rows:"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"header":{"controls":{"copy":{}}},"source":"user_id       pscore   feature_1  feature_2  …\nba890-aced    0.1039   1.007      -9.1333\n93adc-jicae   0.0981   0.093      -0.138\n"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-parameters-2","__idx":18},"children":["Workflow Parameters"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["action_column"]},", ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["reward_column"]},", and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["propensity_column"]}," are not prediction parameters. The input table has no actions or rewards, and the trained policy is frozen at predict time."]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"22%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["input_table"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Source table in dbname.table_name format."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["output_table"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Destination table for function output."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_name"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Yes"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Must match the name used at training time. Preprocessing settings from training, including scaling, imputation, and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["n_predictions"]},", are inherited from the saved model."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["user_id"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Column holding the user identifier. Note this parameter is named ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_column"]}," here, while tune and train use ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_id"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["timestamp_column"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["timestamp"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Event time column."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["exclude_columns"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Pipe-delimited patterns to drop from features, for example ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["col_a|*_raw|temp_*"]},"."]}]}]}]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"output-2","__idx":19},"children":["Output"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Using the predictions."]}," The output table can be activated directly in a Journey, or joined to your customer table so recommendations become attributes available to Master Segment rules. A typical pattern is to join on ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_id"]},", expose the first element of ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["predictions"]}," as a customer attribute, then branch a Journey on that attribute so each user receives the recommended channel or offer. Schedule prediction to match how quickly your context changes; daily is common for e-commerce."]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"15%","data-label":"Field"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Field"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"20%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"65%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["time"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["LONG"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["TD time the prediction was written."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["user_id"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["VARCHAR"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["User identifier carried forward from the input."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["predictions"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["ARRAY<STRING>"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Ordered list of recommended actions, top recommendation first. Currently one action per user."]}]}]}]}]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"header":{"controls":{"copy":{}}},"source":"time        user_id    predictions\n----------  ---------  -------------\n1758004803  23492371   [\"15\"]\n"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":2,"id":"model-types","__idx":20},"children":["Model Types"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," evaluates the candidate set across the four tunable policy families. For most deployments, running tuning and adopting the winning configuration is the recommended path. However, teams may manually override the model type during training if a specific policy class is required. In these cases, we strongly suggest passing the ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_results_table"]}," to inherit validated hyperparameters, while manually tuning model-specific parameters as necessary."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["These sections describe the different model types we currently offer within NBA AI Signals, and also cover ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]},", a model type that does not need tuning."]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"18%","data-label":"Model type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Model type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"14%","data-label":"Category"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Category"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"34%","data-label":"Best for"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Best for"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"34%","data-label":"Main tradeoff"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Main tradeoff"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ipw_learner"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Offline"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Abundant logged data"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Sensitive to propensity accuracy"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ucb"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Online"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Fast baseline with adaptive exploration"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Assumes reward is roughly linear in features"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ts"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Online"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Strong empirical performance"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Slower, due to posterior sampling"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_eps_greedy"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Online"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Simplicity and predictability"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Exploration is uninformed"]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Online, hybrid"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Non-linear structure, large action spaces"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Slowest, outside tuning, needs volume"]}]}]}]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Online models here are simulated: they are trained on logged data rather than interacting with live users."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"inverse-propensity-weighting-learner--ipw_learner-","__idx":21},"children":["Inverse Propensity Weighting Learner (",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ipw_learner"]},")"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Inverse Propensity Weighting Learner. Reweights historical rows by the inverse of how likely the original system was to show each action, then trains a supervised classifier on the reweighted data."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Lightweight and fast, and it works well when logged data is abundant. It is sensitive to propensity accuracy: noisy estimated propensities produce biased results. Supplying a true ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["pscore"]}," column helps most here."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"policy-specific-parameters","__idx":22},"children":["Policy-Specific Parameters"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"22%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["base_classifier"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["string"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["random_forest"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Classifier trained on the reweighted data: ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["random_forest"]}," or ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["logistic_regression"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["rf_n_estimators"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Number of trees. Applies when base_classifier is random_forest. Tuning searches ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["[100, 500]"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["rf_max_depth"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Maximum tree depth. Applies when base_classifier is random_forest. Tuning searches ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["[4, 6, 8, 10]"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["rf_min_sample_split"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Minimum samples to split. Applies when base_classifier is random_forest. Tuning searches ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["[10, 20]"]},". Note the singular spelling; the tuning output column is ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["rf_min_samples_split"]},"."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lr_C"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Regularization strength. Applies when base_classifier is logistic_regression. Distinct from the tuning output's ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["phase1_lr_c"]},", which configures the OPE reward model."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lr_max_iter"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["—"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Iteration cap. Applies when base_classifier is logistic_regression."]}]}]}]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["rf_min_samples_leaf"]}," can be tuned but has no direct train parameter, so it reaches training only through ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_results_table"]},"."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-example","__idx":23},"children":["Workflow Example"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"solution_arguments:\n  model_name: \"nba_retail_ipw_v1\"\n  model_type: \"ipw_learner\"\n  action_column: \"item_id\"\n  reward_column: \"click\"\n  timestamp_column: \"timestamp\"\n  base_classifier: \"random_forest\"\n  propensity_type: \"logistic\"\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"linear-upper-confidence-bound--lin_ucb-","__idx":24},"children":["Linear Upper Confidence Bound (",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ucb"]},")"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Linear Upper Confidence Bound. Maintains a linear reward model per action and picks actions using upper confidence bounds, so uncertainty drives exploration."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Strong theoretical guarantees and adaptive exploration. One of the two fastest options. It assumes reward is approximately linear in your features."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"policy-specific-parameters-1","__idx":25},"children":["Policy-Specific Parameters"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"22%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["epsilon"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Exploration setting. The tuning search space samples ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["[0.01, 0.1, 0.5, 1.0, 5.0]"]},"."]}]}]}]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-example-1","__idx":26},"children":["Workflow Example"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"solution_arguments:\n  model_name: \"nba_retail_ucb_v1\"\n  model_type: \"lin_ucb\"\n  action_column: \"item_id\"\n  reward_column: \"click\"\n  timestamp_column: \"timestamp\"\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"linear-thompson-sampling--lin_ts-","__idx":27},"children":["Linear Thompson Sampling (",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ts"]},")"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Linear Thompson Sampling. Bayesian linear regression with posterior sampling to pick actions."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Excellent empirical performance and natural uncertainty-driven exploration. Slower than the other linear models because of the sampling step. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ts"]}," is a valid ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_type"]}," but is not an option for the default tuning search."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"policy-specific-parameters-2","__idx":28},"children":["Policy-Specific Parameters"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"22%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["epsilon"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Exploration setting for the online policies."]}]}]}]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-example-2","__idx":29},"children":["Workflow Example"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"solution_arguments:\n  model_name: \"nba_retail_ts_v1\"\n  model_type: \"lin_ts\"\n  action_column: \"item_id\"\n  reward_column: \"click\"\n  timestamp_column: \"timestamp\"\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"linear-epsilon-greedy--lin_eps_greedy-","__idx":30},"children":["Linear Epsilon-Greedy (",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_eps_greedy"]},")"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Linear Epsilon-Greedy. A linear reward model plus epsilon-greedy exploration: take the best action with probability (1 - epsilon), pick randomly otherwise."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Simple, predictable, and fast. Exploration is uninformed, so it is less efficient than UCB or Thompson Sampling. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["epsilon"]}," at 0.1 is a reasonable default; 0.0 is pure exploitation and risky, 1.0 is pure random. Like ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ts"]},", this policy is a valid ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["model_type"]}," but is not in the default tuning search space."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"policy-specific-parameters-3","__idx":31},"children":["Policy-Specific Parameters"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"22%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"8%","data-label":"Required"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Required"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["epsilon"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["No"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Probability of exploring instead of exploiting. 0.0 is pure exploitation and risky; 1.0 is pure random. This is the policy where ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["epsilon"]}," carries its literal epsilon-greedy meaning."]}]}]}]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-example-3","__idx":32},"children":["Workflow Example"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"solution_arguments:\n  model_name: \"nba_retail_eps_v1\"\n  model_type: \"lin_eps_greedy\"\n  action_column: \"item_id\"\n  reward_column: \"click\"\n  timestamp_column: \"timestamp\"\n  epsilon: 0.1\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"neurallinucb--neural_lin_ucb-","__idx":33},"children":["NeuralLinUCB (",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]},")"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["NeuralLinUCB. A hybrid: a learned encoder captures non-linear structure in the features, then a linear LinUCB head picks the action in that encoded space."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["It is trained directly through ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]}," and is not part of the ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," search."]}," Reach for it when the linear models underfit and you suspect non-linear structure, or when you have a large action space with useful per-action attributes. Run it as a deliberate comparison against your tuned linear baseline. It is the slowest option and needs more data than the linear policies."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Global vs. per-arm mode"]},", set by ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["per_arm"]},":"]},{"$$mdtype":"Tag","name":"ul","attributes":{},"children":[{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Global (",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["per_arm: false"]},", default)"]}," feeds only the user context to the encoder and keeps a separate LinUCB model per action. Right for a fixed, fairly small action set where every action has plenty of history."]},{"$$mdtype":"Tag","name":"li","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Per-arm (",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["per_arm: true"]},")"]}," concatenates static per-action attributes from ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["arm_feature_columns"]}," to the context and uses one shared model to score every action. Because actions are described by features rather than opaque labels, this handles large or sparse action spaces better and can partially generalize to actions with thin history."]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Warmup."]}," The policy must accumulate roughly ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["warmup_rounds * n_actions"]}," feedback rows to finish warmup and switch to LinUCB. On smaller datasets it may never get there. Set ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["strict_warmup: true"]}," so training fails loudly instead of shipping a model that is still exploring randomly."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"policy-specific-parameters-4","__idx":34},"children":["Policy-Specific Parameters"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb_params"]}]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"24%","data-label":"Parameter"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Parameter"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"14%","data-label":"Type"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Type"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"12%","data-label":"Default"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Default"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"50%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["encoding_dim"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int ≥ 1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["32"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Dimension of the learned encoding feeding the linear UCB head."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["hidden_layer_sizes"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["list of ints"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["[64]"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Encoder MLP hidden layer sizes, for example [64] or [128, 64]."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["per_arm"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["bool"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["false"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Use the per-arm observation model, concatenating arm_feature_columns to the user context."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["learning_rate"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float > 0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.01"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Adam learning rate for encoder and reward-head training."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["weight_decay"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float ≥ 0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Adam weight decay (L2)."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["train_batch_size"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int ≥ 1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["32"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Minibatch size for encoder and reward-head training."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["train_frequency"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int ≥ 1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["50"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["How often, in warmup rounds, the encoder and reward head are trained."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["train_steps_per_update"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int ≥ 1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["32"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["SGD updates performed each time the encoder is trained."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["max_buffer_size"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int ≥ 1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["100000"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Most-recent samples retained in the replay buffer."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["warmup_rounds"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["int ≥ 0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["1000"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Feedback rounds spent in epsilon-greedy warmup before switching to LinUCB."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["strict_warmup"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["bool"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["false"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["When true, training fails if the policy never leaves warmup."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["epsilon_greedy"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float in [0, 1]"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Probability of random exploration during warmup."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lambda_reg"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float > 0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["1.0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Ridge regularization for the LinUCB covariance initialization."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["alpha"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["float ≥ 0"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["0.1"]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["UCB exploration coefficient multiplying the LinUCB confidence interval."]}]}]}]}]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-example-4","__idx":35},"children":["Workflow Example"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"solution_arguments:\n  model_name: \"nba_retail_neural_v1\"\n  model_type: \"neural_lin_ucb\"\n  action_column: \"item_id\"\n  reward_column: \"click\"\n  timestamp_column: \"timestamp\"\n  arm_feature_columns: \"item_price|item_category_*\"\n  neural_lin_ucb_params:\n    per_arm: true\n    encoding_dim: 32\n    hidden_layer_sizes: [128, 64]\n    warmup_rounds: 1000\n    strict_warmup: true\n    alpha: 0.1\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":2,"id":"relevant-topics","__idx":36},"children":["Relevant Topics"]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"reading-tuning-results","__idx":37},"children":["Reading Tuning Results"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["After ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," finishes, pull the best policy's lift against random:"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"sql","header":{"controls":{"copy":{}}},"source":"SELECT\n    phase,\n    estimated_policy_value,\n    ci_lower,\n    ci_upper,\n    lift_vs_random_pct\nFROM your_database.nba_tune_results\nWHERE run_id = '<your_run_id>'\n  AND (phase = 'baseline_random'\n       OR (phase = 'phase2_policy' AND is_best = 'true'));\n","lang":"sql"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Pass condition:"]}," the best policy's ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ci_lower"]}," is above the random baseline's ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["estimated_policy_value"]},", and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lift_vs_random_pct"]}," is comfortably positive. Proceed to training."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["If it fails:"]}," if lift is not positive, or the confidence interval overlaps random, the model may not help. Consider more data, better features, or a different reward definition before deploying. Also check ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["phase2_ess"]},"; a large lift paired with a low effective sample size is a reason to look closer, not a green light. See ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#effective-sample-size-ess-configuration"},"children":["Effective Sample Size (ESS) Configuration"]},"."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Off-policy evaluation is a strong guide, not a substitute for live measurement. A/B test a new policy against the incumbent before full rollout."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"reuse-tuned-hyperparameters","__idx":38},"children":["Reuse Tuned Hyperparameters"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Train with tuning results rather than defaults. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," searches across preprocessing, propensity models, reward models, and OPE estimators to find the best configuration for your data. Training with those values typically produces a meaningfully better policy than generic defaults."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["This is why tuning and training are separate steps. Run the expensive search once on a representative sample, then reuse the validated hyperparameters to train repeatedly without re-tuning."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Point ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_results_table"]}," at the output table from an ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," run. If you do not also set ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tuning_run_id"]},", the most recent run in that table is used. This applies to the four tunable policies; ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," takes its hyperparameters from ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb_params"]}," instead."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":4,"id":"workflow-example-5","__idx":39},"children":["Workflow Example"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"solution_arguments:\n  model_name: \"nba_retail_v1\"\n  model_type: \"lin_eps_greedy\"\n  action_column: \"item_id\"\n  reward_column: \"click\"\n  timestamp_column: \"timestamp\"\n  tuning_results_table: your_database.nba_tune_results\n  tuning_run_id: \"20260815_103422\"\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"effective-sample-size-ess-configuration","__idx":40},"children":["Effective Sample Size (ESS) Configuration"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["With ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["tune_ocv: true"]},", ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," also checks whether each trial's score rests on enough effective data to trust. ",{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["ESS (Effective Sample Size)"]}," measures how many of your historical rows actually count toward evaluating a candidate policy once OPE reweighting is applied."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["OPE estimators like IPW and DR reweight each logged row by how likely the new policy would have taken the action the logging policy actually took. If the new policy behaves quite differently, only a handful of rows carry meaningful weight. You can have 100,000 logged rows where the real signal behind an estimated policy value is closer to a few hundred independent observations. ESS exposes that gap."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ocv_ess_config"]}," controls how strictly Phase 2 enforces the check. Trials below the threshold are dropped during the search."]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"30%","data-label":"Field"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Field"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"70%","data-label":"Description"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Description"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["enabled"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Global toggle for ESS filtering. Default true."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["min_ess_threshold"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Minimum absolute ESS applied across estimators unless overridden. Default 10.0."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ipw"]}," / ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["snipw"]}," / ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["dr"]}," / ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["dros"]}," / ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["dm"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Per-estimator override blocks, each accepting its own enabled, safety_factor, and min_ess_threshold."]}]}]}]}]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Within each block, ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["safety_factor"]}," scales a dynamically calculated threshold based on action coverage and data distribution. Higher values reject more trials. Defaults differ because each estimator tolerates distribution shift differently: ipw 2.0, dr 1.5, dros 1.5, snipw 1.0. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["dm"]}," is disabled by default, since the Direct Method does not reweight by propensity."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Tighten filtering for IPW and DR:"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"ocv_ess_config:\n  enabled: true\n  min_ess_threshold: 20.0\n  ipw:\n    safety_factor: 3.0\n  dr:\n    safety_factor: 2.0\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Give SNIPW a lower bar, since it can work reliably with fewer samples:"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"ocv_ess_config:\n  enabled: true\n  snipw:\n    safety_factor: 1.0\n    min_ess_threshold: 5.0\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Every trial's ESS is written to the tune output as ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["phase2_ess"]},". Leave ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ocv_ess_config"]}," at its defaults in most cases. Revisit it if trials keep getting rejected and tuning is not converging, which usually means thin action coverage; lower ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["min_ess_threshold"]}," or a specific ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["safety_factor"]}," rather than disabling filtering. Raise ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["safety_factor"]}," when you want extra confidence before a production rollout."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"controlling-training-data-size","__idx":41},"children":["Controlling Training Data Size"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Training on very large datasets can exhaust memory. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ipw_learner"]}," and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," are the most demanding, since each fits an additional model (a propensity classifier, or an encoder network) alongside the reward model, building large intermediate arrays that grow with row count."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["The pipeline subsamples automatically. If your dataset exceeds the limit for the chosen model type, it is downsampled using stratified sampling, preserving the distribution across all actions rather than dropping some entirely. Defaults are conservative: ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ipw_learner"]}," and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," cap at 1M rows, the linear models at 5M."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Override per model type:"]},{"$$mdtype":"Tag","name":"CodeBlock","attributes":{"data-language":"yaml","header":{"controls":{"copy":{}}},"source":"max_training_samples_per_model:\n  lin_ucb: 5000000\n  lin_ts: 5000000\n  lin_eps_greedy: 5000000\n  ipw_learner: 1000000\n  neural_lin_ucb: 1000000\n","lang":"yaml"},"children":[]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":["Raising a limit gives the model more data and can improve policy quality, at the cost of longer training times and higher memory usage. Exceeding container memory crashes the job with an out-of-memory error. Be especially careful raising limits for ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ipw_learner"]}," and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]},"."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":2,"id":"faqs","__idx":42},"children":["FAQs"]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"getting-started","__idx":43},"children":["Getting Started"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["What data do I need?"]}," A log of past user-action interactions with a clear reward signal. Each row describes who the user was, what they were shown, what happened, when, and numeric features describing the user at that moment. Categorical features must be encoded before training."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["How do I pick the right reward?"]}," Pick the business outcome you would celebrate if it went up: purchases, clicks, bookings, completed signups. Avoid rewards that are noisy or weakly correlated with value. If your true goal is rare, consider a layered reward: high for purchases, lower for adds-to-cart, zero for nothing. Get this right first; the best model cannot rescue a bad reward."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["What is the difference between ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]},"?"]}," Tuning searches many configurations and reports which policy, preprocessing, and hyperparameters look best for your data. Training fits one chosen configuration on the full dataset and saves a reusable model."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Does tuning use the same data as training?"]}," Yes. Both point at the same user-action interaction table with the same schema. You do not prepare a separate, smaller table for tuning: ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["hyperparam_tune_sample_ratio"]}," subsamples inside the tune run, so by default tuning reads 1% of the rows while training uses all of them."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"data-and-quality","__idx":44},"children":["Data and Quality"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Do I need to provide propensity scores?"]}," No. NBA trains a propensity model to estimate them if ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["pscore"]}," is missing. Including true propensities is still strongly recommended, especially if you logged them during a randomized test."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["What if positive rewards are very rare?"]}," Below roughly 1% of rows, both the propensity and reward models become less reliable, which weakens off-policy evaluation. More data and stronger features help. Validate carefully before acting on results at scale."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["What if an action has very little history?"]}," NBA cannot reliably evaluate whether a policy that chooses it would do well. Aim for at least a few hundred interactions per action. Per-arm ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," handles thin actions better than the alternatives, since it scores them from shared arm features."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Do my features have to be numeric?"]}," Yes. Encode categorical variables using one-hot, target encoding, or embeddings before training. Use ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["exclude_columns"]}," to drop IDs, raw timestamps, and leaky features."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"choosing-a-model","__idx":45},"children":["Choosing a Model"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Which model type should I use?"]}," Run ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_tune"]}," and use what it selects. That is the right answer for most teams."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["When is ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," worth it?"]}," When the linear models underperform and you suspect the relationship between features and response is not linear, or when you have a large action space with useful per-action attributes. It is trained directly through ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["nba_train"]}," rather than selected by tuning, so run it as a deliberate comparison against your tuned baseline."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["What is per-arm mode?"]}," Global mode keeps a separate LinUCB model per action and looks only at user context. Per-arm mode describes each action with its own features and shares one model across all actions. Use per-arm when you have many actions, sparse history per action, or meaningful action metadata."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"operating","__idx":46},"children":["Operating"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["How often should I retrain?"]}," Match the cadence to how fast your user behavior changes. Daily suits fast-moving e-commerce; weekly or monthly is fine for slower categories. Re-run tuning less often, usually when you add actions, change the feature set, or see performance drift. An example cadence could be tuning monthly, training weekly, and predicting daily."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["How do I know my policy is any good?"]}," Check the tuning output: ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ci_lower"]}," above the random baseline's estimated value, and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lift_vs_random_pct"]}," comfortably positive. Then A/B test against your current approach. Off-policy evaluation corrects for the gap between the logging policy and the candidate policy, but the correction is only as good as the propensity and reward models underneath it, so live measurement stays necessary."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["What happens when I add a new action?"]}," The model cannot evaluate or recommend actions it has not seen, so retrain with interactions that include the new one. Per-arm ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," is the closest thing to a workaround today, since it can score a thinly-observed action from its arm features. Full cold-start support via shared-weight hybrid models is on the roadmap."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":3,"id":"scale-and-limits","__idx":47},"children":["Scale and Limits"]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["How does runtime scale?"]}," Training scales with the number of interactions; prediction scales with the number of users. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ucb"]}," and ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_eps_greedy"]}," are fastest, ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["lin_ts"]}," is slower due to posterior sampling, ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["neural_lin_ucb"]}," is slowest. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["ipw_learner"]}," depends on its base classifier. The trained policy is frozen at predict time, so predictions parallelize cleanly; split the input table across parallel API calls for large user bases."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["How large a dataset is supported?"]}," Training subsamples automatically above the per-model-type caps described in ",{"$$mdtype":"Tag","name":"MarkdownLink","attributes":{"href":"#controlling-training-data-size"},"children":["Controlling Training Data Size"]},". Use ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["hyperparam_tune_sample_ratio"]}," to subsample during tuning, then train the winning policy on full data. Detailed benchmarks are being published. Contact your Treasure AI account team for sizing guidance above 100M interactions."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Can NBA recommend more than one action per user?"]}," Not yet. ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["n_predictions"]}," is fixed at 1, so ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["predictions"]}," holds a single top action. Multi-action support, and the ",{"$$mdtype":"Tag","name":"code","attributes":{},"children":["position"]}," input column that enables it, are planned."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Can NBA run in real time?"]}," No. NBA runs as scheduled batch jobs. Sub-daily cadence is supported through frequent workflow runs, but millisecond serving is out of scope."]},{"$$mdtype":"Tag","name":"p","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Can I use negative rewards?"]}," Not with the current online models, which assume non-negative rewards. Use cases needing true negative penalties, such as unsubscribes, require a different policy class. Contact your Treasure AI team if this applies."]},{"$$mdtype":"Tag","name":"Heading","attributes":{"level":2,"id":"glossary","__idx":48},"children":["Glossary"]},{"$$mdtype":"Tag","name":"div","attributes":{"className":"md-table-wrapper"},"children":[{"$$mdtype":"Tag","name":"table","attributes":{"className":"md"},"children":[{"$$mdtype":"Tag","name":"thead","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"th","attributes":{"width":"25%","data-label":"Term"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Term"]}," "]},{"$$mdtype":"Tag","name":"th","attributes":{"width":"75%","data-label":"Definition"},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Definition"]}," "]}]}]},{"$$mdtype":"Tag","name":"tbody","attributes":{},"children":[{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Contextual bandit"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["A model that picks an action for each user (the context) and learns from the reward that follows. A lightweight form of reinforcement learning."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Action"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["One of the options the system can choose, such as an email subject, a coupon, or a send time."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Context"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["The user's features at decision time: age, device, tenure, behavioral signals, anything that helps predict which action will work."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Reward"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["The measured outcome of an action, for example 1 for a click and 0 for no click. The model learns to maximize this."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Policy"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["A learned mapping from context to action. NBA's job is to find one that performs better than your current approach."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Logging policy"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["The rule that picked actions in your historical data, such as random assignment or an old heuristic."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Propensity (pscore)"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["The probability the logging policy chose a given action for a given user. Needed for unbiased evaluation. Estimated if unknown."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Off-Policy Evaluation (OPE)"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["Estimating how well a new policy would perform using data collected under a different policy, without deploying it."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["ESS (Effective Sample Size)"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["How much of your logged data actually contributes to evaluating a new policy. Low ESS means unreliable estimates."]}]},{"$$mdtype":"Tag","name":"tr","attributes":{},"children":[{"$$mdtype":"Tag","name":"td","attributes":{},"children":[{"$$mdtype":"Tag","name":"strong","attributes":{},"children":["Lift vs random"]}]},{"$$mdtype":"Tag","name":"td","attributes":{},"children":["How much better the selected policy is expected to perform than random action assignment."]}]}]}]}]}]},"headings":[{"value":"Next Best Action (NBA) Recommendation","id":"next-best-action-nba-recommendation","depth":1},{"value":"Overview and Use Cases","id":"overview-and-use-cases","depth":2},{"value":"How It Fits with the Other AI Signals","id":"how-it-fits-with-the-other-ai-signals","depth":3},{"value":"Quick Start","id":"quick-start","depth":2},{"value":"Tune","id":"tune","depth":3},{"value":"Train","id":"train","depth":3},{"value":"Predict","id":"predict","depth":3},{"value":"Model Configuration","id":"model-configuration","depth":2},{"value":"nba_tune","id":"nba_tune","depth":3},{"value":"Data Preparation","id":"data-preparation","depth":4},{"value":"Workflow Parameters","id":"workflow-parameters","depth":4},{"value":"Output","id":"output","depth":4},{"value":"nba_train","id":"nba_train","depth":3},{"value":"Data Preparation","id":"data-preparation-1","depth":4},{"value":"Workflow Parameters","id":"workflow-parameters-1","depth":4},{"value":"Output","id":"output-1","depth":4},{"value":"nba_predict","id":"nba_predict","depth":3},{"value":"Data Preparation","id":"data-preparation-2","depth":4},{"value":"Workflow Parameters","id":"workflow-parameters-2","depth":4},{"value":"Output","id":"output-2","depth":4},{"value":"Model Types","id":"model-types","depth":2},{"value":"Inverse Propensity Weighting Learner ( ipw_learner )","id":"inverse-propensity-weighting-learner--ipw_learner-","depth":3},{"value":"Policy-Specific Parameters","id":"policy-specific-parameters","depth":4},{"value":"Workflow Example","id":"workflow-example","depth":4},{"value":"Linear Upper Confidence Bound ( lin_ucb )","id":"linear-upper-confidence-bound--lin_ucb-","depth":3},{"value":"Policy-Specific Parameters","id":"policy-specific-parameters-1","depth":4},{"value":"Workflow Example","id":"workflow-example-1","depth":4},{"value":"Linear Thompson Sampling ( lin_ts )","id":"linear-thompson-sampling--lin_ts-","depth":3},{"value":"Policy-Specific Parameters","id":"policy-specific-parameters-2","depth":4},{"value":"Workflow Example","id":"workflow-example-2","depth":4},{"value":"Linear Epsilon-Greedy ( lin_eps_greedy )","id":"linear-epsilon-greedy--lin_eps_greedy-","depth":3},{"value":"Policy-Specific Parameters","id":"policy-specific-parameters-3","depth":4},{"value":"Workflow Example","id":"workflow-example-3","depth":4},{"value":"NeuralLinUCB ( neural_lin_ucb )","id":"neurallinucb--neural_lin_ucb-","depth":3},{"value":"Policy-Specific Parameters","id":"policy-specific-parameters-4","depth":4},{"value":"Workflow Example","id":"workflow-example-4","depth":4},{"value":"Relevant Topics","id":"relevant-topics","depth":2},{"value":"Reading Tuning Results","id":"reading-tuning-results","depth":3},{"value":"Reuse Tuned Hyperparameters","id":"reuse-tuned-hyperparameters","depth":3},{"value":"Workflow Example","id":"workflow-example-5","depth":4},{"value":"Effective Sample Size (ESS) Configuration","id":"effective-sample-size-ess-configuration","depth":3},{"value":"Controlling Training Data Size","id":"controlling-training-data-size","depth":3},{"value":"FAQs","id":"faqs","depth":2},{"value":"Getting Started","id":"getting-started","depth":3},{"value":"Data and Quality","id":"data-and-quality","depth":3},{"value":"Choosing a Model","id":"choosing-a-model","depth":3},{"value":"Operating","id":"operating","depth":3},{"value":"Scale and Limits","id":"scale-and-limits","depth":3},{"value":"Glossary","id":"glossary","depth":2}],"frontmatter":{"seo":{"title":"Next Best Action (NBA) Recommendation","description":"Predict the optimal next action, channel, send time, or offer for every customer, built on AI Signals, Treasure AI's ML platform."}},"lastModified":"2026-08-20T18:43:14.000Z","pagePropGetterError":{"message":"","name":""}},"slug":"/ja/products/customer-data-platform/machine-learning/ai-signals/nba-ai-signals","userData":{"isAuthenticated":false,"teams":["anonymous"]},"isPublic":true}