weeyuga-benchmarks-public/submissions/EXAMPLE/mac-m1-8gb/run-00000000-0000-0000-0000-000000000000/run.jsonl

{"type":"meta","benchmark_run_id":"00000000-0000-0000-0000-000000000000","harness_version":"public-1","started_at_utc":"2026-05-12T14:32:11Z","host_hostname_short":"alices-mbp","load_avg_start":[1.2,1.4,1.6],"target_url":"http://127.0.0.1:11434","cell_id_prefix":"mac-m1:ollama","submitter_handle":"EXAMPLE","device_tag":"mac-m1-8gb","execution_shape":"per-model-block","phases_planned":["hello","5q","20q"],"models_planned":["qwen3.5:0.8b"],"canonical_options":{"temperature":0.1,"num_ctx":4096,"num_predict":2048},"canonical_options_effective":{"temperature":0.1,"num_ctx":4096,"num_predict":2048},"timeout_seconds":360,"platform_system":"Darwin","platform_release":"23.5.0","python_version":"3.12.4"}
{"type":"call","ts_utc":"2026-05-12T14:32:13Z","cell_id":"mac-m1:ollama:qwen3.5:0.8b","model":"qwen3.5:0.8b","phase":"hello","question_id":"hello_check","run_idx":0,"duration_seconds":1.847,"prompt_tokens":17,"completion_tokens":42,"tokens_per_second":22.74,"finish_reason":"stop","status_code":200,"response_chars":167,"response_preview":"Hello! Of course, I'd be happy to help. What can I assist you with today? Whether it's a question, a task, or just a chat, I'm here to help.","required_markers":[],"markers_hit":[],"marker_hit_rate":null,"format_rule":"","format_ok":null,"usable_answer":true,"error":null}
{"type":"call","ts_utc":"2026-05-12T14:32:21Z","cell_id":"mac-m1:ollama:qwen3.5:0.8b","model":"qwen3.5:0.8b","phase":"5q","question_id":"Q1","run_idx":0,"duration_seconds":4.21,"prompt_tokens":58,"completion_tokens":94,"tokens_per_second":22.32,"finish_reason":"stop","status_code":200,"response_chars":312,"response_preview":"#!/usr/bin/env bash\nset -euo pipefail\n\nif [[ ! -d \"$1\" ]]; then\n  echo \"err: $1 not a directory\" >&2\n  exit 1\nfi\n\nfor f in \"$1\"/*.log; do\n  [[ -e \"$f\" ]] || continue\n  gzip -k \"$f\"\ndone","required_markers":["gzip","#!/usr/bin/env bash"],"markers_hit":["gzip","#!/usr/bin/env bash"],"marker_hit_rate":1.0,"format_rule":"bash_code","format_ok":true,"usable_answer":true,"error":null}
{"type":"call","ts_utc":"2026-05-12T14:33:48Z","cell_id":"mac-m1:ollama:qwen3.5:0.8b","model":"qwen3.5:0.8b","phase":"20q","question_id":"Q01","run_idx":0,"duration_seconds":9.612,"prompt_tokens":142,"completion_tokens":201,"tokens_per_second":20.91,"finish_reason":"stop","status_code":200,"response_chars":784,"response_preview":"def is_valid_ipv4(addr: str) -> bool:\n    parts = addr.split('.')\n    if len(parts) != 4:\n        return False\n    for p in parts:\n        if not p.isdigit():\n            return False\n        n = int(p)","required_markers":["is_valid_ipv4","def test_"],"markers_hit":["is_valid_ipv4","def test_"],"marker_hit_rate":1.0,"format_rule":"python_code","format_ok":true,"usable_answer":true,"error":null}
{"type":"call","ts_utc":"2026-05-12T14:36:42Z","cell_id":"mac-m1:ollama:qwen3.5:0.8b","model":"qwen3.5:0.8b","phase":"20q","question_id":"Q14","run_idx":13,"duration_seconds":42.118,"prompt_tokens":189,"completion_tokens":312,"tokens_per_second":7.41,"finish_reason":"stop","status_code":200,"response_chars":1240,"response_preview":"To debug this MyBoard auth issue, the triage should focus on…","required_markers":["/auth/login","/auth/me","myboard:post-login-redirect","tenant-missing"],"markers_hit":["/auth/login","/auth/me","tenant-missing"],"marker_hit_rate":0.75,"format_rule":"json_dict","format_ok":false,"usable_answer":true,"error":null}
{"type":"footer","ts_utc":"2026-05-12T14:48:03Z","finished_at_utc":"2026-05-12T14:48:03Z","load_avg_end":[1.6,1.5,1.6]}