← Files JinkōARCHIVED FILE
skills/jinko-trial/evals/evals.json
2.54 KB · Oct 2, 2026 · 00:29 UTC
{
"skill_name": "jinko-trial",
"evals": [
{
"id": 1,
"prompt": "Create a minimum Jinkō trial from model cm-123 that saves Drug and run it only if sanity passes.",
"expected_output": "Creates a simple output set from Drug, creates the trial from the model, checks sanity/status before run, uses --apply/--run gates, and polls with wait_until_completed."
},
{
"id": 2,
"prompt": "Find a completed trial in my project, print its summary, and download timeseries and scalar results as pandas dataframes.",
"expected_output": "Uses client.list_trials, filters status == completed, calls trial.results.summary, trial.output_ids, trial.results.timeseries(...).to_dataframe, and trial.results.scalars(...).to_dataframe."
},
{
"id": 3,
"prompt": "Why can't I launch this trial when my data table armScope does not match the protocol arms?",
"expected_output": "Explains trial sanity constraints, specifically that data-table armScope values must match protocol arms, and recommends fixing data table or protocol before launch."
},
{
"id": 4,
"prompt": "My advanced output set validated fine standalone, but the Jinkō UI shows 'The following advanced outputs have errors' when I try to launch trial tr-abc. What's going on and how do I check it from Python?",
"expected_output": "Explains that standalone validate_scoring_formula()/diagnostics only check the scoring design in isolation and cannot catch trial-context failures, then calls trial.sanity() (a raw dict, not a typed object) and inspects report['scorings']['sanity']['errors'] (ADVANCED_OUTPUTS_ERRORS) and report['scorings']['sanity']['componentsSanity'] to find the failing component, rather than relying on jinko-output-set's standalone checks."
},
{
"id": 5,
"prompt": "The output set on my current trial is wrong. Fix this same run without leaving a trail of trial_v2 and output_set_v2 assets.",
"expected_output": "Inspects and repairs the existing assets where possible, re-runs trial.sanity(), and only creates replacements if an independent scenario or preservation requirement makes reuse unsafe."
},
{
"id": 6,
"prompt": "Attach data table dt-123 to a new trial, but only if it is confirmed valid for fitness.",
"expected_output": "Requires metadata.public.validForFitnessFunction to be exactly True, rejects false or missing values, and passes the DataTable through client.create_trial(..., data_tables=[table]) rather than constructing raw project-item references."
}
]
}
SHA-256: db3677f530445cbdddf4ae06223ca1f238dfcf4091e8201bd4c6dae2d000b1a4