init: seed framework reference content from agent-runtimes main repo
This commit is contained in:
14
model-registry/airouter-qwen3.yaml
Normal file
14
model-registry/airouter-qwen3.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: airouter-qwen3
|
||||
endpoint: airouter-qwen3
|
||||
scores:
|
||||
complexity: 8
|
||||
test_pass_rate: 7
|
||||
creativity: 7
|
||||
spec_adherence: 7
|
||||
cost_efficiency: 10
|
||||
context_utilisation: 8
|
||||
cost_per_1k_tokens:
|
||||
input: 0.0
|
||||
output: 0.0
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/claude-haiku-4.yaml
Normal file
14
model-registry/claude-haiku-4.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: claude-haiku-4
|
||||
endpoint: claude-haiku-4
|
||||
scores:
|
||||
complexity: 5
|
||||
test_pass_rate: 7
|
||||
creativity: 6
|
||||
spec_adherence: 8
|
||||
cost_efficiency: 9
|
||||
context_utilisation: 6
|
||||
cost_per_1k_tokens:
|
||||
input: 0.0008
|
||||
output: 0.004
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/claude-opus-4.yaml
Normal file
14
model-registry/claude-opus-4.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: claude-opus-4
|
||||
endpoint: claude-opus-4
|
||||
scores:
|
||||
complexity: 10
|
||||
test_pass_rate: 9
|
||||
creativity: 10
|
||||
spec_adherence: 9
|
||||
cost_efficiency: 3
|
||||
context_utilisation: 10
|
||||
cost_per_1k_tokens:
|
||||
input: 0.015
|
||||
output: 0.075
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/claude-sonnet-4.yaml
Normal file
14
model-registry/claude-sonnet-4.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: claude-sonnet-4
|
||||
endpoint: claude-sonnet-4
|
||||
scores:
|
||||
complexity: 8
|
||||
test_pass_rate: 9
|
||||
creativity: 8
|
||||
spec_adherence: 9
|
||||
cost_efficiency: 6
|
||||
context_utilisation: 8
|
||||
cost_per_1k_tokens:
|
||||
input: 0.003
|
||||
output: 0.015
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/github-models-gpt4o.yaml
Normal file
14
model-registry/github-models-gpt4o.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: github-models-gpt4o
|
||||
endpoint: github-models-gpt4o
|
||||
scores:
|
||||
complexity: 8
|
||||
test_pass_rate: 8
|
||||
creativity: 8
|
||||
spec_adherence: 8
|
||||
cost_efficiency: 7
|
||||
context_utilisation: 8
|
||||
cost_per_1k_tokens:
|
||||
input: 0.0
|
||||
output: 0.0
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/gpt-4o-openai.yaml
Normal file
14
model-registry/gpt-4o-openai.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: gpt-4o-openai
|
||||
endpoint: gpt-4o-openai
|
||||
scores:
|
||||
complexity: 8
|
||||
test_pass_rate: 8
|
||||
creativity: 8
|
||||
spec_adherence: 8
|
||||
cost_efficiency: 7
|
||||
context_utilisation: 8
|
||||
cost_per_1k_tokens:
|
||||
input: 0.0025
|
||||
output: 0.01
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/opencode-zen-sonnet.yaml
Normal file
14
model-registry/opencode-zen-sonnet.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: opencode-zen-sonnet
|
||||
endpoint: opencode-zen-sonnet
|
||||
scores:
|
||||
complexity: 8
|
||||
test_pass_rate: 9
|
||||
creativity: 8
|
||||
spec_adherence: 9
|
||||
cost_efficiency: 6
|
||||
context_utilisation: 8
|
||||
cost_per_1k_tokens:
|
||||
input: 0.003
|
||||
output: 0.015
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/openrouter-sonnet.yaml
Normal file
14
model-registry/openrouter-sonnet.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: openrouter-sonnet
|
||||
endpoint: openrouter-sonnet
|
||||
scores:
|
||||
complexity: 8
|
||||
test_pass_rate: 9
|
||||
creativity: 8
|
||||
spec_adherence: 9
|
||||
cost_efficiency: 6
|
||||
context_utilisation: 8
|
||||
cost_per_1k_tokens:
|
||||
input: 0.003
|
||||
output: 0.015
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
14
model-registry/qwen3-235b.yaml
Normal file
14
model-registry/qwen3-235b.yaml
Normal file
@@ -0,0 +1,14 @@
|
||||
name: qwen3-235b
|
||||
endpoint: qwen3-235b-vllm
|
||||
scores:
|
||||
complexity: 8
|
||||
test_pass_rate: 8
|
||||
creativity: 7
|
||||
spec_adherence: 8
|
||||
cost_efficiency: 8
|
||||
context_utilisation: 7
|
||||
cost_per_1k_tokens:
|
||||
input: 0.0002
|
||||
output: 0.0006
|
||||
source: benchmark
|
||||
harness_version: null
|
||||
Reference in New Issue
Block a user