first commit
Security: Sync from Public / sync-from-public (push) Has been cancelled
Test: Benchmark Nightly / build (push) Has been cancelled
Test: Benchmark Nightly / Notify Cats on failure (push) Has been cancelled
CI: Python / Checks (push) Has been cancelled
Test: Evals Python / Workflow Comparison Python (push) Has been cancelled
Util: Check Docs URLs / check-docs-urls (push) Has been cancelled
Test: Visual Storybook / Cloudflare Pages (push) Has been cancelled
Test: E2E Performance / build-and-test-performance (push) Has been cancelled
Test: Workflows Nightly / Run Workflow Tests (push) Has been cancelled
Util: Cleanup CI Docker Images / Delete stale CI images (push) Has been cancelled
Test: Benchmark Destroy Env / build (push) Has been cancelled
Util: Update Node Popularity / update-popularity (push) Has been cancelled
Test: E2E Coverage Weekly / Coverage Tests (push) Has been cancelled
Security: Sync from Public / sync-from-public (push) Has been cancelled
Test: Benchmark Nightly / build (push) Has been cancelled
Test: Benchmark Nightly / Notify Cats on failure (push) Has been cancelled
CI: Python / Checks (push) Has been cancelled
Test: Evals Python / Workflow Comparison Python (push) Has been cancelled
Util: Check Docs URLs / check-docs-urls (push) Has been cancelled
Test: Visual Storybook / Cloudflare Pages (push) Has been cancelled
Test: E2E Performance / build-and-test-performance (push) Has been cancelled
Test: Workflows Nightly / Run Workflow Tests (push) Has been cancelled
Util: Cleanup CI Docker Images / Delete stale CI images (push) Has been cancelled
Test: Benchmark Destroy Env / build (push) Has been cancelled
Util: Update Node Popularity / update-popularity (push) Has been cancelled
Test: E2E Coverage Weekly / Coverage Tests (push) Has been cancelled
This commit is contained in:
@@ -0,0 +1,46 @@
|
||||
// ============================================================================
|
||||
// Agent Types
|
||||
// ============================================================================
|
||||
|
||||
export const AGENT_TYPES = {
|
||||
CODE_BUILDER: 'code-builder',
|
||||
MULTI_AGENT: 'multi-agent',
|
||||
} as const;
|
||||
|
||||
// ============================================================================
|
||||
// Evaluation Type Identifiers
|
||||
// ============================================================================
|
||||
|
||||
export const EVAL_TYPES = {
|
||||
PAIRWISE_LOCAL: 'pairwise-local',
|
||||
PAIRWISE_LANGSMITH: 'pairwise-langsmith',
|
||||
LANGSMITH: 'langsmith-evals',
|
||||
} as const;
|
||||
|
||||
export const EVAL_USERS = {
|
||||
PAIRWISE_LOCAL: 'pairwise-local-user',
|
||||
LANGSMITH: 'langsmith-eval-user',
|
||||
} as const;
|
||||
|
||||
export const TRACEABLE_NAMES = {
|
||||
PAIRWISE_EVALUATION: 'pairwise_evaluation',
|
||||
WORKFLOW_GENERATION: 'workflow_generation',
|
||||
} as const;
|
||||
|
||||
// ============================================================================
|
||||
// Default Values
|
||||
// ============================================================================
|
||||
|
||||
export const DEFAULTS = {
|
||||
NUM_JUDGES: 3,
|
||||
EXPERIMENT_NAME: 'pairwise-evals',
|
||||
LLM_JUDGE_EXPERIMENT_NAME: 'workflow-builder-evaluation',
|
||||
CONCURRENCY: 5,
|
||||
REPETITIONS: 1,
|
||||
/** Per-operation timeout (generation / evaluator) */
|
||||
TIMEOUT_MS: 20 * 60 * 1000,
|
||||
DATASET_NAME: 'notion-pairwise-workflows',
|
||||
FEATURE_FLAGS: {
|
||||
templateExamples: false,
|
||||
},
|
||||
} as const;
|
||||
Reference in New Issue
Block a user