{"data":{"service":{"name":"hatchery","version":"0.1.0","projection_version":"hatchery-query/v1","revision_schema_version":"hatchery-revision/v1","documentation_url":"https://hatchery.clavia.ai/docs","openapi_url":"https://hatchery.clavia.ai/api/v1/openapi.json"},"principal":{"principal_id":"anonymous","principal_kind":"capability","grant_id":null},"actions":[],"workspaces":[],"tables":{"attempts":[{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"run_id","label":"Run","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"revision_id","label":"Revision","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"slot_id","label":"Slot","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"column_key","label":"System","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"column_label","label":"System name","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"item_key","label":"Task","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"attempt_key","label":"Attempt","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"attempt_id","label":"Source attempt","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"work_type","label":"Work type","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"category","label":"Category","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"execution_status","label":"Execution","type":"enum","values":["completed","timed_out","failed","incomplete","unknown"]},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"submitted","label":"Submitted","type":"boolean"},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"graded","label":"Graded","type":"boolean"},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"assessable","label":"Assessable","type":"boolean"},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"eligible","label":"Counted in metrics","type":"boolean"},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"strict_pass","label":"Strict pass","type":"boolean"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":"fraction","format":"percent","groupable":false,"field":"substance_fraction","label":"Substance","type":"number"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":"fraction","format":"percent","groupable":false,"field":"form_fraction","label":"Form","type":"number"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":null,"format":"number","groupable":false,"field":"substance_passed","label":"Substance passed","type":"integer"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":null,"format":"number","groupable":false,"field":"substance_total","label":"Substance counted","type":"integer"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":"currency","format":"currency","groupable":false,"field":"resolved_cost","label":"Resolved cost","type":"number"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":"currency","format":"currency","groupable":false,"field":"reported_cost","label":"Reported cost","type":"number"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":"tokens","format":"number","groupable":false,"field":"input_tokens","label":"Input tokens","type":"integer"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":"tokens","format":"number","groupable":false,"field":"output_tokens","label":"Output tokens","type":"integer"},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"excluded","label":"Excluded","type":"boolean"}],"grades":[{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"run_id","label":"Run","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"revision_id","label":"Revision","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"slot_id","label":"Slot","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"column_key","label":"System","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"item_key","label":"Task","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"attempt_key","label":"Attempt","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"attempt_id","label":"Source attempt","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"grade_id","label":"Grade","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"evaluation_id","label":"Evaluation","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"criterion_key","label":"Criterion","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"axis","label":"Axis","type":"enum","values":["substance","form"]},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"verdict","label":"Verdict","type":"enum","values":["pass","fail"]},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"assessment_state","label":"Assessment","type":"enum","values":["settled","unsettled","not_assessable","missing"]},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"origin","label":"Origin","type":"enum","values":["machine","operator_correction","external_analysis"]},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"retained","label":"Retained","type":"boolean"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"judge_model","label":"Judge","type":"string"},{"operators":["contains","eq"],"unit":null,"format":"text","groupable":true,"field":"rationale","label":"Rationale","type":"text","description":"The judge's recorded reasoning. Readable only by a grant carrying the judge_rationale visibility class; otherwise null, with rationale_withheld true."},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"rationale_withheld","label":"Rationale withheld","type":"boolean","description":"True when a rationale exists and this grant may not read it. It distinguishes a withheld rationale from one the judge never recorded."}],"artifacts":[{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"artifact_id","label":"Artifact","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"name","label":"Name","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"role","label":"Role","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"media_type","label":"Media type","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"digest","label":"Digest","type":"string"},{"operators":["eq","in","gt","gte","lt","lte"],"unit":"bytes","format":"number","groupable":false,"field":"size_bytes","label":"Size","type":"integer"},{"operators":["eq"],"unit":null,"format":"text","groupable":true,"field":"sealed","label":"Sealed","type":"boolean"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"target_kind","label":"Target kind","type":"string"},{"operators":["eq","in","contains"],"unit":null,"format":"text","groupable":true,"field":"target_id","label":"Target","type":"string"},{"operators":["contains","eq"],"unit":null,"format":"text","groupable":false,"field":"text","label":"Extracted text","type":"text"}]},"operators":["eq","in","gt","gte","lt","lte","contains"],"metrics":[{"id":"form_attempt_mean","label":"Form mean","description":"The mean, over eligible attempts, of the share of that attempt's form criteria that passed.","implementation":"legalbench:1851e94efd831f413d00ce676e1798ac7caef9ea#form","producer_field":"form","unit":"fraction","axis":"form","higher_is_better":true,"aggregation":"mean","population":"selected_attempts","numerator":"Sum of per-attempt form fractions.","denominator":"Eligible attempts that have form criteria.","eligibility":"Same as the attempt pass rate, and the attempt has a form roster. A column graded without the form axis reports this metric as unavailable.","zero_denominator":"null_with_coverage","required_attempt_keys":null,"display":{"format":"percent","digits":1}},{"id":"reported_cost","label":"Reported cost","description":"The cost the producer recorded for the selected attempts, retained unmodified.","implementation":"reported/v1","producer_field":"reported_cost","unit":"currency","axis":null,"higher_is_better":false,"aggregation":"sum","population":"selected_attempts","numerator":"Sum of reported attempt costs.","denominator":"Selected attempts that report a cost.","eligibility":"The attempt is selected and reports a cost.","zero_denominator":"null_with_coverage","required_attempt_keys":null,"display":{"format":"currency","digits":3}},{"id":"resolved_cost","label":"Resolved cost","description":"The cost of the selected attempts, recalculated from captured usage against the selected rate sheet.","implementation":"cost/v1","producer_field":null,"unit":"currency","axis":null,"higher_is_better":false,"aggregation":"sum","population":"selected_attempts","numerator":"Sum of resolved attempt costs.","denominator":"Selected attempts with an available cost.","eligibility":"The attempt is selected and its usage prices against the rate sheet. Grading eligibility does not apply: a failed execution that spent tokens still cost money.","zero_denominator":"null_with_coverage","required_attempt_keys":null,"display":{"format":"currency","digits":3}},{"id":"source_expenditure","label":"Source expenditure","description":"Everything the retained source records say was spent, including superseded attempts and judge calls. This is spend, not the cost of the displayed benchmark.","implementation":"reported/v1","producer_field":null,"unit":"currency","axis":null,"higher_is_better":false,"aggregation":"sum","population":"source_records","numerator":"Sum of reported costs over source attempts and judge calls, deduplicated by source identity.","denominator":"Source records that report a cost.","eligibility":"Every retained source record in scope, selected or not.","zero_denominator":"null_with_coverage","required_attempt_keys":null,"display":{"format":"currency","digits":3}},{"id":"substance_attempt_mean","label":"Attempt mean","description":"The mean, over eligible attempts, of the share of that attempt's substance criteria that passed.","implementation":"legalbench:1851e94efd831f413d00ce676e1798ac7caef9ea#attempt_macro_average","producer_field":"attempt_macro_average","unit":"fraction","axis":"substance","higher_is_better":true,"aggregation":"mean","population":"selected_attempts","numerator":"Sum of per-attempt substance fractions.","denominator":"Eligible attempts with at least one counted criterion.","eligibility":"Same as the attempt pass rate.","zero_denominator":"null_with_coverage","required_attempt_keys":null,"display":{"format":"percent","digits":1}},{"id":"substance_attempt_pass_rate","label":"Attempt pass rate","description":"Attempts whose every substance criterion passed, over the attempts eligible to be scored.","implementation":"legalbench:1851e94efd831f413d00ce676e1798ac7caef9ea#pass_1","producer_field":"pass_1","unit":"fraction","axis":"substance","higher_is_better":true,"aggregation":"ratio","population":"selected_attempts","numerator":"Eligible attempts with a strict substance pass.","denominator":"Eligible attempts.","eligibility":"The attempt is selected, graded, assessable, and its roster is complete.","zero_denominator":"null_with_coverage","required_attempt_keys":null,"display":{"format":"percent","digits":1}},{"id":"substance_criterion_pass_rate","label":"Criterion pass rate","description":"Passed substance criteria over all substance criteria across eligible attempts.","implementation":"legalbench:1851e94efd831f413d00ce676e1798ac7caef9ea#checkpoint_pass_rate","producer_field":"checkpoint_pass_rate","unit":"fraction","axis":"substance","higher_is_better":true,"aggregation":"ratio","population":"selected_attempts","numerator":"Substance criteria with a settled passing verdict.","denominator":"Substance criteria counted for eligible attempts, including unsettled ones.","eligibility":"Same as the attempt pass rate.","zero_denominator":"null_with_coverage","required_attempt_keys":null,"display":{"format":"percent","digits":1}},{"id":"substance_task_pass_rate","label":"Task pass rate","description":"Tasks whose required attempts all passed, over the tasks whose required attempts are all eligible.","implementation":"legalbench:1851e94efd831f413d00ce676e1798ac7caef9ea#pass_2","producer_field":"pass_2","unit":"fraction","axis":"substance","higher_is_better":true,"aggregation":"ratio","population":"selected_attempts","numerator":"Paired tasks whose every required attempt passed strictly.","denominator":"Tasks whose every required attempt is eligible.","eligibility":"Every attempt key named by the revision's paired_attempt_keys is eligible for that task.","zero_denominator":"null_with_coverage","required_attempt_keys":["attempt-001","attempt-002"],"display":{"format":"percent","digits":1}}],"field_paths":[{"path":"columns.{column_key}.label","description":"The display name of an evaluated system.","value_type":"string","dependents":[]},{"path":"columns.{column_key}.configuration_artifact_id","description":"The captured configuration of an evaluated system. Changing it declares that the column is a different configuration, which comparison checks then report.","value_type":"string","dependents":["comparison"]},{"path":"columns.{column_key}.configuration_digest","description":"The configuration digest attempts are checked against. Set it explicitly to accept a rerun under a changed configuration.","value_type":"string","dependents":["comparison","attempt_compatibility"]},{"path":"items.{item_key}.category","description":"The task category a row groups under.","value_type":"string","dependents":["group_by:category"]},{"path":"items.{item_key}.work_type","description":"The declared work type. It selects the form criteria roster, so changing it changes which form criteria apply.","value_type":"string","dependents":["form_attempt_mean","grading_roster"]},{"path":"items.{item_key}.title","description":"The display title of a task.","value_type":"string","dependents":[]},{"path":"slots.{slot_id}.usage.input_tokens","description":"The input token count a cost calculation bills.","value_type":"integer","dependents":["resolved_cost"]},{"path":"slots.{slot_id}.usage.output_tokens","description":"The output token count a cost calculation bills.","value_type":"integer","dependents":["resolved_cost"]},{"path":"slots.{slot_id}.usage.cached_input_tokens","description":"The cached input token count a cost calculation rebates.","value_type":"integer","dependents":["resolved_cost"]},{"path":"slots.{slot_id}.input_ref.input_digest","description":"The input this slot expects an attempt to have run against. Set it explicitly to accept an attempt that ran on changed task input.","value_type":"string","dependents":["attempt_compatibility","comparison"]}],"policy_paths":[{"path":"grading_mix","description":"require_compatible allows one judge contract per selected grading set; explicit permits a deliberate mixture and reports it.","value_type":"string","dependents":["grading_compatibility"]},{"path":"ungraded_attempts","description":"Whether an attempt with no grades counts in the metrics.","value_type":"string","dependents":["all_substance_metrics"]},{"path":"unassessable_attempts","description":"Whether an attempt the judge could not assess counts.","value_type":"string","dependents":["all_substance_metrics"]},{"path":"unsettled_criteria","description":"not_passed keeps an unsettled criterion in the denominator with no credit; exclude drops it.","value_type":"string","dependents":["all_substance_metrics"]},{"path":"partial_grading","description":"What to do with an attempt whose roster has a criterion with no grade.","value_type":"string","dependents":["all_substance_metrics"]},{"path":"paired_attempt_keys","description":"The attempt keys the paired task metric requires.","value_type":"string","dependents":["substance_task_pass_rate"]},{"path":"cost.rate_sheet_artifact_id","description":"The captured provider rate sheet costs are calculated from.","value_type":"string","dependents":["resolved_cost"]},{"path":"cost.calculation_version","description":"The cost calculator version.","value_type":"string","dependents":["resolved_cost"]},{"path":"cost.currency","description":"The currency resolved costs are reported in.","value_type":"string","dependents":["resolved_cost"]},{"path":"cost.billing_category","description":"Which billing category to price against when a rate sheet holds more than one for a model.","value_type":"string","dependents":["resolved_cost"]},{"path":"cost.provider","description":"Which provider to price against when a rate sheet holds more than one for a model.","value_type":"string","dependents":["resolved_cost"]},{"path":"cost.missing_cached_rate","description":"What to do when an attempt reports cached tokens and the rate sheet declares no cached rate.","value_type":"string","dependents":["resolved_cost"]},{"path":"cost.conversion_artifact_id","description":"The captured conversion source a cost crossing currencies requires.","value_type":"string","dependents":["resolved_cost"]},{"path":"comparison.alignment","description":"How rows from different columns are matched.","value_type":"string","dependents":["comparison"]},{"path":"comparison.configuration","description":"Whether incompatible configurations block a comparison or are shown as differences.","value_type":"string","dependents":["comparison"]}],"change_kinds":["select_attempt","select_grades","set_field","use_projection","set_policy","exclude_slot","attach"],"artifact_roles":["original_archive","collection_freeze","candidate_result","deliverable","trace_events","grading_plan","grading_cell","judge_call","producer_summary","dataset_item","configuration","rate_sheet","currency_conversion","extraction","external_analysis","note","report_result","export_manifest","attachment"],"default_policy":{"grading_mix":"require_compatible","ungraded_attempts":"exclude","unassessable_attempts":"exclude","unsettled_criteria":"not_passed","partial_grading":"exclude","paired_attempt_keys":["attempt-001","attempt-002"],"cost":{"rate_sheet_artifact_id":null,"calculation_version":"cost/v1","currency":"USD","conversion_artifact_id":null,"billing_category":null,"provider":null,"missing_cached_rate":"unavailable"},"comparison":{"alignment":"same_input","configuration":"require_compatible"}},"limits":{"default_page_size":50,"max_page_size":200,"default_text_preview_characters":4096,"max_upload_bytes":134217728,"max_changes_per_revision":5000,"max_query_selection":20,"max_export_rows":1000000}},"links":{"canonical":"https://hatchery.clavia.ai/api/v1/context","api":"https://hatchery.clavia.ai/api/v1/context","ui":"https://hatchery.clavia.ai/docs"},"meta":{"max_scope_records":200000,"visibility_classes":[],"expires_at":null,"resource_scope":{"kind":"workspace"}}}