{"ok":true,"count":17,"benchmarks":[{"id":"geoanalystbench-gisclaw","canonical_name":"GeoAnalystBench / GISclaw","aliases":["GeoAnalystBench","GISclaw"],"paper_ref":"https://arxiv.org/abs/2603.26845","repo_ref":"https://github.com/GeoDS/GeoAnalystBench","dataset_ref":"https://github.com/GeoDS/GeoAnalystBench/blob/master/dataset/GeoAnalystBench.csv","leaderboard_ref":null,"publication_date":"2026-03","latest_update":"2026-06-12","task_count":50,"task_families":["vector analysis","raster processing","table operations","visualisation","network analysis","ArcGIS layer/project handling"],"difficulty":"mixed (3-10 subtasks per task, average 5.8 steps)","geography":"varied (real-world geoprocessing tasks from GIS platforms, software, online tutorials, and academic literature)","modality":"mixed","input_types":["vector","raster","tabular","netCDF","network","ArcGIS layer/project files"],"output_types":["Python code","PNG images","GeoJSON","rendered map outputs"],"evaluates":"agent","execution_environment":"persistent Python sandbox with open-source GIS libraries","allowed_tool_surface":"GeoPandas, rasterio, scipy, scikit-learn, matplotlib (open-source only; no ArcPy)","ground_truth_method":"expert-designed reference Python solution per task; expected outputs (PNG/GeoJSON)","verifier_method":"L1 API F1, L2 reasoning similarity, L3 output verification (vector geometry diff via shapely, raster pixel diff via numpy, tabular value diff via pandas). Binary pass/fail.","scoring_dimensions":["L1 API F1","L2 reasoning similarity","L3 output verification"],"aggregation_formula":"binary pass/fail per task (all three layers must pass)","trials":"600 (6 models x 2 architectures x 50 tasks)","variance_reporting":"not explicitly reported","data_license":"pending verification","code_license":"pending verification (open-source repository)","contamination_concerns":"tasks are expert-designed and public via CSV; contamination possible if models trained on the benchmark data","reproducibility_status":"high (GeoAnalystBench is public with data and reference scripts)","what_it_proves":"Code execution success on greenfield GIS analysis. SA (single-agent ReAct) dominates DA (Plan-Execute-Replan) for strong models: DeepSeek-V3.2 96% SA vs 32% DA. Error recovery adds approximately 8 percentage points.","what_it_does_not_prove":"Production deployment, guardrail compliance, migration nuance, over-engineering, documentation quality, cloud-platform correctness, data-volume awareness.","axis_status":"downloaded","axis_evidence":"extracted-full/ at .external/geoanalystbench/, 50 task folders with datasets + reference .py + expected outputs. GeoAnalystBench.csv. SHA-256 of full zip fdaea3ae439169174278fc22f9a3bfa0c60811c3e7f3b462c6ab4306b95e52a3."},{"id":"geobenchx","canonical_name":"GeoBenchX","aliases":["GeoBenchX"],"paper_ref":"https://arxiv.org/abs/2503.18129","repo_ref":"https://github.com/solirinai/geobenchx","dataset_ref":"Google Drive data bundle (linked from repository)","leaderboard_ref":null,"publication_date":"2025-03","latest_update":"pending verification","task_count":202,"task_families":["geospatial tool-use","mixed reasoning","geospatial API calls"],"difficulty":"mixed","geography":"varied","modality":"mixed","input_types":["tool calls","geospatial APIs","mixed reasoning prompts"],"output_types":["tool traces","reference solutions"],"evaluates":"model+tools","execution_environment":"pending verification","allowed_tool_surface":"geospatial APIs, tool calls","ground_truth_method":"reference solutions (tool traces, not final numeric or geospatial gold outputs)","verifier_method":"LLM-as-judge plus reference-solution comparison","scoring_dimensions":["LLM judge score","reference-solution comparison"],"aggregation_formula":"pending verification","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"reference solutions are tool traces, not final numeric or geospatial gold outputs; LLM-judge scoring is non-deterministic","reproducibility_status":"medium (repository public, data via Google Drive bundle)","what_it_proves":"Geospatial tool-use and agent task completion with reference-solution comparison.","what_it_does_not_prove":"Reference solutions are tool traces, not final numeric or geospatial gold outputs. Does not prove production deployment, guardrail compliance, or deterministic output verification.","axis_status":"downloaded","axis_evidence":"/root/axis-spatial/.external/geobenchx/ — upstream repo + geobenchx-data-download/ (159 files: GeoData, StatData, source notes, BibTeX). SHA-256 afb80eecdb731425743efaf9c916a59f54db17567ca5576e83989223cb5595e9."},{"id":"geo-bench-2","canonical_name":"GEO-Bench-2","aliases":["GEO-Bench-2"],"paper_ref":"pending verification","repo_ref":"https://github.com/The-AI-Alliance/GEO-Bench-2","dataset_ref":"Official Hugging Face dataset versions with labels and targets","leaderboard_ref":"https://github.com/The-AI-Alliance/GEO-Bench-2-Leaderboard","publication_date":"pending verification","latest_update":"pending verification","task_count":17,"task_count_note":"17 datasets (downstream tasks), not individual task instances","task_families":["EO foundation-model downstream tasks: classification, segmentation, change detection, regression, retrieval"],"difficulty":"varied by dataset","geography":"global (multi-sensor EO coverage)","modality":"mixed","input_types":["raster EO","multi-sensor satellite imagery"],"output_types":["classification labels","segmentation masks","regression values","change detection maps"],"evaluates":"model","execution_environment":"model evaluation (foundation model downstream; not agent execution)","allowed_tool_surface":"foundation model inference","ground_truth_method":"labels and targets in official Hugging Face dataset versions","verifier_method":"accuracy, F1, Jaccard or IoU, RMSE depending on dataset","scoring_dimensions":["accuracy","F1","Jaccard/IoU","RMSE"],"aggregation_formula":"per-dataset metrics; no universal aggregate across 17 datasets","trials":"pending verification","variance_reporting":"pending verification","data_license":"varies by dataset (Hugging Face)","code_license":"pending verification","contamination_concerns":"public datasets; contamination possible if foundation models trained on them","reproducibility_status":"high (Hugging Face datasets with pinned labels and targets)","what_it_proves":"EO foundation-model downstream task quality across 17 datasets spanning multiple sensors.","what_it_does_not_prove":"Agent or tool-use execution. Does not test code generation, workflow planning, or artifact production.","axis_status":"not-downloaded","axis_evidence":"Not downloaded. Planned as model-selection benchmark and EO or GFM comparator. Pin HF dataset versions, licences, selected split, and cache refs before use."},{"id":"geo-bench-2-leaderboard","canonical_name":"GEO-Bench-2 Leaderboard","aliases":["GEO-Bench-2 Leaderboard"],"paper_ref":null,"repo_ref":"https://github.com/The-AI-Alliance/GEO-Bench-2-Leaderboard","dataset_ref":null,"leaderboard_ref":"https://github.com/The-AI-Alliance/GEO-Bench-2-Leaderboard","publication_date":"pending verification","latest_update":"pending verification","task_count":null,"task_count_note":"not a task benchmark; aggregated leaderboard over submitted metrics","task_families":["aggregated leaderboard over submitted metrics"],"difficulty":"n/a","geography":"n/a","modality":"mixed","input_types":["submission CSV","model metadata"],"output_types":["leaderboard rankings"],"evaluates":"model","execution_environment":"n/a (submission-based registry)","allowed_tool_surface":"n/a","ground_truth_method":"depends on underlying GEO-Bench-2 datasets","verifier_method":"normalised leaderboard aggregation over submitted metrics","scoring_dimensions":["submitted metrics"],"aggregation_formula":"normalised leaderboard aggregation","trials":"n/a","variance_reporting":"n/a","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"depends on underlying GEO-Bench-2 datasets and TerraTorch or iterate workflow","reproducibility_status":"medium (depends on submission integrity and upstream dataset versioning)","what_it_proves":"Aggregated public model-performance registry for EO foundation models.","what_it_does_not_prove":"Not a data source or golden task. Depends on GEO-Bench-2 official datasets and TerraTorch or iterate workflow.","axis_status":"not-downloaded","axis_evidence":"Not downloaded. Comparator for baseline model selection and reporting, not a data source."},{"id":"geobench-vlm","canonical_name":"GEOBench-VLM","aliases":["GEOBench-VLM","geo-bench-vlm"],"paper_ref":"pending verification","repo_ref":"https://github.com/the-ai-alliance/geo-bench-vlm","dataset_ref":"Hugging Face dataset with manually verified instructions and answer options","leaderboard_ref":null,"publication_date":"pending verification","latest_update":"pending verification","task_count":"pending verification","task_families":["geospatial VLM multiple-choice question answering"],"difficulty":"pending verification","geography":"pending verification","modality":"mixed","modality_note":"remote-sensing imagery plus language; specific sensor coverage pending verification","input_types":["remote-sensing imagery","language instructions"],"output_types":["multiple-choice answers"],"evaluates":"model","execution_environment":"VLM inference (repo includes inference scripts)","allowed_tool_surface":"VLM model inference","ground_truth_method":"manually verified instructions and answer options","verifier_method":"MCQ accuracy","scoring_dimensions":["MCQ accuracy"],"aggregation_formula":"MCQ accuracy","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"public Hugging Face dataset; contamination possible if VLMs trained on it","reproducibility_status":"medium (Hugging Face dataset with inference scripts; GPU, model, and provider posture must be explicit)","what_it_proves":"Geospatial VLM multiple-choice question-answering accuracy on remote-sensing imagery.","what_it_does_not_prove":"Not a primary spatial verifier. MCQ accuracy does not prove code execution, artifact production, or workflow correctness.","axis_status":"not-downloaded","axis_evidence":"Not downloaded. VLM sidecar benchmark and possible small pinned image-understanding seed."},{"id":"geommbench-geommagent","canonical_name":"GeoMMBench / GeoMMAgent","aliases":["GeoMMBench","GeoMMAgent","GeoMM-AGI"],"paper_ref":"https://arxiv.org/abs/2604.08896","repo_ref":"https://github.com/Shihao-Cheng/GeoMMAgent","dataset_ref":"https://huggingface.co/datasets/AR-X/GeoMMBench","leaderboard_ref":null,"publication_date":"2026-04","latest_update":"2026-05-02","task_count":1053,"task_count_note":"1,053 rows on HuggingFace (37-question public validation split + 1,017-question private test split); 303 MB","task_families":["scene classification","object detection","change detection","spectral analysis","spatial reasoning","terrain characterisation","GNSS pseudorange calculations"],"difficulty":"expert-level (derived from geoscience curricula, not crowd-sourced)","geography":"varied (remote sensing, photogrammetry, GIS, GNSS disciplines)","modality":"mixed","modality_note":"optical, SAR, hyperspectral, LiDAR, DEM, thermal across remote sensing, photogrammetry, GIS, and GNSS","input_types":["image","text (question)","multiple-choice options A-D"],"output_types":["multiple-choice answer (A/B/C/D)"],"evaluates":"model","evaluates_note":"GeoMMBench evaluates VLMs; GeoMMAgent is a separate multi-agent framework (coordinator plus perception, search, reasoning, and self-evaluation agents)","execution_environment":"VLM inference (zero-shot, 36+ VLMs tested); GeoMMAgent runs YOLO11 and DeepLabV3+ perception models locally","allowed_tool_surface":"VLM model inference; GeoMMAgent adds YOLO11 (scene classification, oriented bounding box detection), DeepLabV3+ (semantic segmentation), web search, GME image-text similarity filtering","ground_truth_method":"expert-derived multiple-choice questions with verified answers","verifier_method":"MCQ accuracy; optional self-evaluation dimensions (logic, spatial reasoning, domain validity, accuracy)","scoring_dimensions":["MCQ accuracy","optional self-eval: logic, spatial reasoning, domain validity, accuracy"],"aggregation_formula":"MCQ accuracy","trials":"36+ VLMs tested under zero-shot conditions","variance_reporting":"pending verification (per-model accuracy numbers not published in public README at time of analysis)","data_license":"CC BY 4.0 (dataset)","code_license":"Apache 2.0 (GeoMMAgent framework)","contamination_concerns":"public Hugging Face dataset; 37-question public validation split is small and may yield high variance in external reproductions","reproducibility_status":"medium (dataset public on HuggingFace; per-model accuracy in paper PDF; arXiv identifier confirmed as 2604.08896)","what_it_proves":"Expert multimodal geoscience and remote-sensing MCQ accuracy across six sensor modalities and four geoscience disciplines. CVPR 2026 Highlight.","what_it_does_not_prove":"Closed-form MCQ does not prove execution. A model scoring 90% may still fail to produce a cloud-free NDVI composite or execute a spatial workflow.","axis_status":"not-downloaded","axis_evidence":"Not downloaded. VLM sidecar benchmark, multimodal model-selection comparator, and possible small pinned image-understanding seed."},{"id":"gabench-geoagentbench","canonical_name":"GABench / GeoAgentBench","aliases":["GABench","GeoAgentBench","GeoAgentBench"],"paper_ref":"https://arxiv.org/abs/2604.13888","repo_ref":"https://github.com/geox-lab/GABench","dataset_ref":"https://github.com/geox-lab/GABench (Git LFS required for task data and ground-truth assets)","leaderboard_ref":null,"publication_date":"2026-04","latest_update":"2026-06-12","task_count":53,"task_count_note":"53 tasks, 117 atomic GIS tools, 6 GIS domains, average 6.7 tool calls per task, maximum 17","task_families":["buffer analysis","overlay","topology repair","raster algebra","geocoding","coordinate reprojection","map rendering"],"difficulty":"mixed (single-tool tasks through chained 4+ tool workflows)","geography":"varied (real GIS workflows across 6 GIS domains)","modality":"mixed","input_types":["vector","raster","tabular","netCDF","graph or network","rendered map outputs"],"output_types":["rendered GT maps","data products","toolchain execution traces"],"evaluates":"complete-system","execution_environment":"tool-augmented agent execution (117 atomic GIS tools)","allowed_tool_surface":"117 atomic GIS tools across 6 GIS domains","ground_truth_method":"generated and verified GT maps and data products; ground-truth tool call sequences per task","verifier_method":"TAO (Tool Accuracy Overall), TIO (Tool Invocation Overall), TEM (Tool Execution Metric), PEA (Parameter Execution Accuracy) with last-attempt alignment and file-existence checks; VLM-as-judge map comparison; execution-efficiency metrics","scoring_dimensions":["TAO","TIO","TEM","PEA","VLM map comparison","execution efficiency"],"aggregation_formula":"composite of TAO, TIO, TEM, PEA with last-attempt alignment; file-existence checks gate pass; VLM judge for map comparison","trials":"pending verification (7 LLMs evaluated; GPT-4o and Claude 3.5 Sonnet perform best)","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"public repo uses Git LFS; normal clone without LFS leaves core task and data files as pointers, blocking reproduction","reproducibility_status":"medium (repository public but LFS assets required; task CSV, GT maps, data, dependency lock, output isolation, and VLM judge posture must be imported)","what_it_proves":"Tool-use execution, parameter inference, and map-product verification. GPT-4o approximately 72%, Claude 3.5 Sonnet approximately 68%; chained 4+ tool tasks drop to approximately 45%. PEA measures parameter-level accuracy within tolerance.","what_it_does_not_prove":"Production deployment, guardrails, cloud-platform correctness, data-volume awareness.","axis_status":"not-downloaded","axis_evidence":"Not downloaded (LFS assets needed). Strong comparator for execution, parameter inference, and map-product verification. Not first-ten ready until LFS assets, task CSV, GT maps, data, dependency lock, output isolation, and VLM judge posture are imported."},{"id":"mapqa","canonical_name":"MapQA","aliases":["MapQA"],"paper_ref":"https://arxiv.org/abs/2503.07871","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"2025-03","latest_update":"pending verification","task_count":10421,"task_count_note":"10,421 questions over 50 US cities","task_families":["proximity","containment","routing","aggregation","comparison"],"difficulty":"mixed (single-hop through multi-hop 3+ step reasoning)","geography":"50 US cities","modality":"vector","input_types":["OSM geometries","POI metadata","routing queries"],"output_types":["answer (text)"],"evaluates":"model","evaluates_note":"evaluates LLM reasoning (model-only) and tool-augmented LLM reasoning (model+tools)","execution_environment":"LLM inference; optional tool-augmented mode with geocoding API, routing API, and spatial query tool","allowed_tool_surface":"geocoding API, routing API, spatial query tool (tool-augmented mode)","ground_truth_method":"derived from OSM geometries and POI metadata with known spatial relationships","verifier_method":"accuracy (exact match against ground-truth answer)","scoring_dimensions":["accuracy"],"aggregation_formula":"accuracy (percentage correct); per-category breakdown (proximity, containment, routing, aggregation, comparison)","trials":"pending verification (GPT-4, GPT-3.5, Gemini Pro, LLaMA-3-70B tested; plus tool-augmented GPT-4)","variance_reporting":"pending verification","data_license":"pending verification (underlying OSM data is ODbL)","code_license":"pending verification","contamination_concerns":"questions derived from public OSM data; contamination possible if models trained on similar QA pairs","reproducibility_status":"medium (methodology public; dataset availability pending verification)","what_it_proves":"Spatial reasoning gap: GPT-4 64.2% overall, drops to 41.3% on multi-hop (3+ steps). Tool augmentation adds +17.5pp (81.7%). Routing: GPT-4 22.4% vs tool-augmented 76.8%. Tools add huge lift.","what_it_does_not_prove":"Code execution or artifact correctness. Does not test workflow generation, spatial data processing, or map production.","axis_status":"not-downloaded","axis_evidence":"Not downloaded. Spatial reasoning and tool-augmentation comparator."},{"id":"thinkgeo","canonical_name":"ThinkGeo","aliases":["ThinkGeo"],"paper_ref":"pending verification","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"2025","latest_update":"pending verification","task_count":"pending verification","task_families":["tool-augmented remote sensing"],"difficulty":"pending verification","geography":"pending verification","modality":"mixed","modality_note":"remote-sensing focused; specific sensor coverage pending verification","input_types":["remote-sensing data","tool calls"],"output_types":["pending verification"],"evaluates":"model+tools","execution_environment":"pending verification","allowed_tool_surface":"remote-sensing tools (specific surface pending verification)","ground_truth_method":"pending verification","verifier_method":"step correctness and final correctness","scoring_dimensions":["step correctness","final correctness"],"aggregation_formula":"pending verification","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"pending verification","reproducibility_status":"pending verification","what_it_proves":"Tool-augmented remote sensing benchmark with step-level and final correctness scoring.","what_it_does_not_prove":"Source custody, artifact lineage, or production acceptance states. Benchmark scoring, not a production-readiness contract.","axis_status":"none","axis_evidence":"Not downloaded. Identified in Axis geospatial verification gap research as a tool-augmented RS benchmark with step and final correctness. Deeper fields pending verification pending source review."},{"id":"earth-bench-earth-agent","canonical_name":"Earth-Bench / Earth-Agent","aliases":["Earth-Bench","Earth-Agent"],"paper_ref":"pending verification","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"2026","latest_update":"pending verification","task_count":"pending verification","task_families":["spatiotemporal reasoning","MCP-style EO tools"],"difficulty":"pending verification","geography":"pending verification","modality":"mixed","modality_note":"EO focused; specific sensor coverage pending verification","input_types":["EO data","MCP-style tool calls"],"output_types":["trajectory outcomes","final outcomes"],"evaluates":"model+tools","execution_environment":"MCP-style EO tool environment (pending verification)","allowed_tool_surface":"MCP-style EO tools (specific surface pending verification)","ground_truth_method":"pending verification","verifier_method":"trajectory and final outcome evaluation for spatiotemporal reasoning","scoring_dimensions":["trajectory evaluation","final outcome evaluation"],"aggregation_formula":"pending verification","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"pending verification","reproducibility_status":"pending verification (described as a research framework, not a product)","what_it_proves":"Trajectory and final outcome evaluation for spatiotemporal reasoning using MCP-style EO tools.","what_it_does_not_prove":"Product proof objects for user-produced spatial work. Research framework, not a production verification contract.","axis_status":"none","axis_evidence":"Not downloaded. Identified in Axis geospatial verification gap research as an MCP-style EO tool and trajectory or final outcome evaluation framework for spatiotemporal reasoning. Deeper fields pending verification pending source review."},{"id":"openearth-bench-openearthagent","canonical_name":"OpenEarth-Bench / OpenEarthAgent","aliases":["OpenEarth-Bench","OpenEarthAgent","OpenEarth Agent"],"paper_ref":"pending verification","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"2026","latest_update":"pending verification","task_count":"pending verification","task_families":["unified tool registry","structured tool calls","GIS, spectral, GeoTIFF tool trajectories"],"difficulty":"pending verification","geography":"pending verification","modality":"mixed","modality_note":"GIS, spectral, GeoTIFF; specific sensor coverage pending verification","input_types":["GIS data","spectral data","GeoTIFF rasters","structured tool calls"],"output_types":["tool trajectories"],"evaluates":"model+tools","execution_environment":"unified tool registry with structured tool calls (pending verification)","allowed_tool_surface":"unified tool registry spanning GIS, spectral, and GeoTIFF operations (specific surface pending verification)","ground_truth_method":"pending verification","verifier_method":"structured tool-call and tool-trajectory evaluation","scoring_dimensions":["tool-call correctness","tool-trajectory evaluation"],"aggregation_formula":"pending verification","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"pending verification","reproducibility_status":"pending verification (described as a training or evaluation framework, not a product)","what_it_proves":"Unified tool registry with structured tool calls and GIS, spectral, and GeoTIFF tool trajectories.","what_it_does_not_prove":"Backend-owned verification records and projection readiness. Training or evaluation framework, not a production verification contract.","axis_status":"none","axis_evidence":"Not downloaded. Identified in Axis geospatial verification gap research as a unified tool registry with structured tool calls covering GIS, spectral, and GeoTIFF tool trajectories. Deeper fields pending verification pending source review."},{"id":"terrabench","canonical_name":"TerraBench","aliases":["TerraBench"],"paper_ref":"pending verification","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"pending verification","latest_update":"pending verification","task_count":"pending verification","task_families":["pending verification"],"difficulty":"pending verification","geography":"pending verification","modality":"mixed","modality_note":"EO benchmark family; specific sensor coverage pending verification","input_types":["pending verification"],"output_types":["pending verification"],"evaluates":"pending verification","execution_environment":"pending verification","allowed_tool_surface":"pending verification","ground_truth_method":"pending verification","verifier_method":"pending verification","scoring_dimensions":["pending verification"],"aggregation_formula":"pending verification","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"pending verification","reproducibility_status":"pending verification","what_it_proves":"pending verification","what_it_does_not_prove":"pending verification","axis_status":"none","axis_evidence":"Not downloaded. Listed in the GeoAI Analysis product plan under TerraBench and related EO benchmark families. No detailed data found in GBrain at time of registry construction. All fields pending verification pending source review."},{"id":"rsrcc","canonical_name":"RSRCC","aliases":["RSRCC","Remote Sensing Regional Change Comprehension"],"paper_ref":"https://arxiv.org/abs/2604.20623","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"2026-04","latest_update":"2026-05-02","task_count":126000,"task_count_note":"approximately 126,000 questions for fine-grained semantic reasoning about localised changes in remote sensing image pairs","task_families":["remote sensing change comprehension","fine-grained spatial localisation","bi-temporal image pair reasoning"],"difficulty":"mixed (global binary change detection through fine-grained sub-region identification)","geography":"varied (bi-temporal remote sensing image pairs)","modality":"optical","modality_note":"remote sensing imagery; specific sensor coverage pending verification","input_types":["bi-temporal remote sensing image pairs","natural language questions"],"output_types":["fine-grained change descriptions with spatial localisation"],"evaluates":"model","execution_environment":"MLLM or VLM inference","allowed_tool_surface":"multimodal large language model inference","ground_truth_method":"hierarchical semi-supervised curation pipeline with Best-of-N ranking as final ambiguity-resolution stage","verifier_method":"accuracy on fine-grained questions versus global binary questions","scoring_dimensions":["fine-grained change identification accuracy","global binary change detection accuracy"],"aggregation_formula":"accuracy; difficulty gap measured between fine-grained and global binary questions","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"large-scale public benchmark; contamination possible if MLLMs trained on it","reproducibility_status":"medium (methodology public; dataset and code availability pending verification)","what_it_proves":"Fine-grained semantic reasoning about localised changes in remote sensing image pairs. Top models achieve approximately 60-65% on fine-grained questions versus 80%+ on global binary questions. The difficulty gap is largest for spectrally similar changes.","what_it_does_not_prove":"Code execution, workflow generation, or artifact production. Does not test agent tool-use or production deployment.","axis_status":"none","axis_evidence":"Not downloaded. Remote-sensing change-comprehension benchmark. The annotation pipeline may be adaptable for generating training data for fine-tuned summarisation models."},{"id":"urbansarfloods","canonical_name":"UrbanSARFloods","aliases":["UrbanSARFloods","Sentinel-1 Flood Benchmark"],"paper_ref":"https://arxiv.org/abs/2406.04111","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"2024-06","latest_update":"pending verification","task_count":8879,"task_count_note":"8,879 chips covering 807,500 square kilometres across 5 continents, with 20 land-cover classes at Sentinel-1 SLC resolution","task_families":["SAR flood mapping","pixel-level flood or no-flood classification","land-cover-stratified evaluation"],"difficulty":"mixed (class-imbalanced; urban flooding is hardest due to double-bounce SAR artefacts)","geography":"5 continents, 807,500 square kilometres","modality":"SAR","input_types":["Sentinel-1 SLC chips"],"output_types":["pixel-level flood or no-flood labels"],"evaluates":"model","execution_environment":"deep learning model evaluation (pixel-level segmentation)","allowed_tool_surface":"deep learning baselines (specific models pending verification)","ground_truth_method":"pixel-level flood or no-flood labels stratified by land-cover class and continent","verifier_method":"F1 score per chip, stratified by land-cover class and urban versus open-area","scoring_dimensions":["F1 score","urban chip F1","open-area chip F1","class-imbalance robustness"],"aggregation_formula":"F1 score per chip; stratified by land-cover class and continent","trials":"pending verification (several deep learning baselines evaluated)","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"large-scale benchmark; contamination possible if flood detection models trained on it","reproducibility_status":"medium (methodology public; dataset availability pending verification)","what_it_proves":"SAR flood detection is hard in urban environments: best baselines achieve 0.52-0.65 F1 on urban chips versus 0.78-0.85 on open-area chips. Imbalanced class distribution (under 5% flooded pixels in typical scenes) is the dominant challenge.","what_it_does_not_prove":"Agent execution, workflow generation, or production deployment. Does not test code generation or tool use.","axis_status":"none","axis_evidence":"Not downloaded. Related EO benchmark family. Used as a reference standard in the Bearlin v1 benchmark spec for SAR flood task acceptance criteria (flooded km2 within 25% of UrbanSARFloods reference)."},{"id":"geoai-agency-primitives","canonical_name":"GeoAI Agency Primitives","aliases":["GeoAI Agency Primitives","Agency Primitives Benchmark"],"paper_ref":"https://arxiv.org/abs/2604.01869","repo_ref":"pending verification","dataset_ref":"pending verification","leaderboard_ref":null,"publication_date":"2026-04","latest_update":"pending verification","task_count":"pending verification","task_count_note":"benchmark measures analyst productivity across eight task types; task count pending verification","task_families":["navigation","perception","geo-referenced memory","action planning","spatial reasoning","tool composition","multi-step coordination","uncertainty quantification","human-in-the-loop delegation"],"difficulty":"mixed (eight task types)","geography":"pending verification","modality":"mixed","input_types":["GIS workflow tasks","geospatial data"],"output_types":["analyst productivity metrics","primitive implementation scores"],"evaluates":"agent","execution_environment":"GeoAI assistant with varying subsets of nine agency primitives","allowed_tool_surface":"GIS tools (specific surface pending verification)","ground_truth_method":"analyst productivity baseline comparison across eight task types","verifier_method":"productivity gain measurement (percentage improvement versus unassisted baseline)","scoring_dimensions":["analyst productivity","per-primitive contribution (action planning, tool composition, geo-referenced memory, perception, spatial reasoning, uncertainty quantification)"],"aggregation_formula":"productivity gain percentage versus unassisted baseline; per-primitive ablation","trials":"pending verification","variance_reporting":"pending verification","data_license":"pending verification","code_license":"pending verification","contamination_concerns":"pending verification","reproducibility_status":"pending verification","what_it_proves":"Agents implementing all nine primitives improve analyst productivity by 60-80%. Highest-value primitives: action planning (40% gain), tool composition (25%), geo-referenced memory (15%). Uncertainty quantification is rarely implemented but expected in high-stakes domains.","what_it_does_not_prove":"Code execution correctness, artifact production, or spatial verification. Productivity gain does not equal output correctness.","axis_status":"none","axis_evidence":"Not downloaded. Spatial-reasoning and capability framework beyond MapQA. Nine primitive taxonomy used as a capability framework reference in the GeoAI Analysis product plan."},{"id":"axis-migration-benchmark","canonical_name":"Axis Migration Benchmark (v5/v6)","aliases":["Axis v5 Benchmark","Axis v6 Benchmark","Axis Migration Benchmark"],"paper_ref":null,"repo_ref":"internal (tasks/benchmark/v6/)","dataset_ref":"internal (ExpDis Japan ArcPy workflow, 4000+ LOC)","leaderboard_ref":null,"publication_date":"2026-04","latest_update":"2026-04-02","task_count":8,"task_count_note":"v5: 1 workflow (ExpDis Japan, 4000+ LOC ArcPy), 5 analysis questions; v6: 3 cases (simple_buffer T1, buffer+clip+area T2, site_suitability T3 on AWS), 4 configs per case","task_families":["ArcPy-to-open-source migration planning","migration code generation","guardrail compliance","platform-specific deployment"],"difficulty":"mixed (T1 single-transform through T3 multi-script site suitability with AWS platform constraints)","geography":"Japan (ExpDis production workflow)","modality":"mixed","input_types":["ArcPy source code","production workflow context","migration requirements"],"output_types":["migration analysis documents (v5)","generated Python code with guardrail checks (v6)"],"evaluates":"model","evaluates_note":"v5 evaluates text analysis quality; v6 evaluates code generation and execution. Both use single-model evaluation, not multi-agent pipelines.","execution_environment":"v5: text analysis (no code execution); v6: code generation with AST parse, guardrail scan, arcpy import check, DRY_RUN guard, and pathlib usage verification","allowed_tool_surface":"v5: LLM text generation; v6: LLM code generation with automated guardrail scanning","ground_truth_method":"v5: 7-dimension rubric (Structural, Domain, Errors, Guardrails, Nuance, Docs, Over-engineering); v6: AST parse success, guardrail scan pass, arcpy absence, DRY_RUN guard presence, pathlib usage check","verifier_method":"v5: rubric scoring (0-10 per dimension, 70 total); v6: automated code verification (AST parse, guardrail scan, import check, DRY_RUN, pathlib)","scoring_dimensions":["Structural","Domain","Errors","Guardrails","Nuance","Docs","Over-engineering (v5); AST parse, guardrail scan, arcpy check, DRY_RUN, pathlib (v6)"],"aggregation_formula":"v5: continuous 0-10 per dimension, 70 total; v6: binary pass or fail per case per config","trials":"v5: 16 models x 1 architecture = 16 runs; v6: 8 Bedrock models x 4 configs x 3 cases = 96 runs","variance_reporting":"v5: no variance (N=1 per model); v6: no variance (N=1 per config per model per case)","data_license":"internal (proprietary)","code_license":"internal (proprietary)","contamination_concerns":"proprietary workflow; no public contamination risk","reproducibility_status":"medium (evaluator is deterministic; workflow is proprietary; v6 results on real Bedrock APIs inside staging container on Hetzner)","what_it_proves":"v5: production deployment, guardrail compliance, migration nuance, over-engineering, documentation quality. v6: code execution success (SA 96%, SA_RETRY 100%, PA 100%, FULL 100%). Confirmed GISclaw SA greater than DA finding: SA produces same quality as FULL for 10-19x fewer LLM calls.","what_it_does_not_prove":"Breadth: 1 workflow (v5) or 3 cases (v6) versus 50 tasks in GeoAnalystBench. No repeated trials. No gold-standard output comparison (v6). Not generalisable to non-migration tasks.","axis_status":"reproduced","axis_evidence":"Reproduced internally. v5 results: tasks/benchmark/v5/ leaderboard with 16 models. v6 results: tasks/benchmark/v6/results/v6_report.md and v6_results.json with 96 runs. Full results in docs/research/geospatial-verification/benchmark-meta-analysis-axis-vs-gisclaw."},{"id":"axis-geoworkbench","canonical_name":"GeoWorkBench","aliases":["GeoWorkBench","Bearlin v1"],"paper_ref":null,"repo_ref":"internal (planned)","dataset_ref":"internal (planned)","leaderboard_ref":null,"publication_date":null,"latest_update":"2026-06-12","task_count":30,"task_count_note":"v0.1 spec drafted (Bearlin v1): 30 tasks across 9 categories. Reference values still TBD. Not yet run.","task_families":["CRS and coordinate diagnosis","vector buffer, overlay, join, geometry repair","raster clip, reprojection, index, zonal statistics","mixed raster or vector suitability analysis","spatial SQL and geometry queries","data discovery and STAC selection","Earth Engine and EO workflow execution","map artifact and report creation","ArcPy-to-open-source migration","multi-file workflow reconstruction","negative controls for invalid, missing, or impossible data","long-running tasks, retries, and remediation","runtime portability and deployment-readiness","multi-stage production workflow packaging"],"difficulty":"mixed (T0 negative controls through T3 hard multi-step)","geography":"global (Munich, Bavaria, Bangladesh, Pakistan, Mozambique, Greece, Sahel, Iowa, California, Lake Tahoe, Lake Mead, Greenland, Mars (negative control), Antarctica (negative control))","modality":"mixed","input_types":["natural-language requests","AOI definitions","date ranges","STAC collection references"],"output_types":["data-map-layer","data-image","text reports","refusal messages (negative controls)"],"evaluates":"complete-system","execution_environment":"E2B sandbox (axis-geo-raster template) with geopandas, rasterio, shapely, pyproj; Cloudflare Worker for run intake and verification","allowed_tool_surface":"geopandas, rasterio, shapely, pyproj, GEE, STAC, OSM (pending finalisation)","ground_truth_method":"per-task quantitative criteria (value constraints) and expert acceptance criteria; reference values from Copernicus, UNOSAT, UrbanSARFloods, USGS (currently TBD)","verifier_method":"4 binary axes (A1 artefact present, A2 values plausible, A3 code executed, A4 expert agrees) plus 2 continuous metrics (time, cost); deterministic verification first, LLM judge as side metric only","scoring_dimensions":["accepted task completion","parameter execution accuracy","tool-use correctness","artifact existence","spatial correctness","numeric correctness","policy and guardrail compliance","evidence and provenance quality","failure honesty","recovery rate","cost and latency","tool-call and retry efficiency","reproducibility","production readiness"],"aggregation_formula":"task pass = all 4 axes pass; per-category breakdown; correct-refusal rate; zero-pixel-success rate (target 0%)","trials":"planned (minimum 3 per configuration for stochastic configurations)","variance_reporting":"planned (mean plus or minus standard deviation)","data_license":"internal (proprietary tasks; underlying data licences separate)","code_license":"internal (proprietary)","contamination_concerns":"private task set with 24 public and 6 private hold-out tasks to prevent overfit","reproducibility_status":"planned (v0.1 spec drafted; reference values TBD; not yet run)","what_it_proves":"Planned: production-readiness including CRS correctness, guardrail compliance, false-success resistance, correct refusal, recovery after failure, reproducibility, cost per accepted task, and deployment suitability.","what_it_does_not_prove":"Not yet built. v0.1 spec drafted with 30 tasks but reference values pending. Cannot prove anything until tasks are run with pinned data, deterministic checks, and repeated trials.","axis_status":"none","axis_evidence":"Planned, not yet built. v0.1 spec (Bearlin v1) drafted at docs/research/eval-observability/bearlin-v1-benchmark-spec with 30 tasks across 9 categories (ndvi-analysis, sar-flood, zonal-ndvi-statistics, ndwi-time-series, multi-date-composite, multi-step-analysis, scene-retrieval, edge-case, multi-turn). Reference values from Copernicus, UNOSAT, UrbanSARFloods, and USGS are TBD. 24 public and 6 private hold-out tasks."}]}