{"id":797,"slug":"huggingfacem4--finevisionmax","name":"FineVisionMax","author":"HuggingFaceM4","description":"\n\t\n\t\t\n\t\tFine Vision\n\t\n\n\nFineVision is a massive collection of datasets with 17.3M images, 24.3M samples, 88.9M turns, and 9.5B answer tokens, designed for training state-of-the-art open Vision-Language-Models.\nMore detail can be found in the blog post: https://huggingface.co/spaces/HuggingFaceM4/FineVision\nThe version in this repository concatenated all the configs in the original dataset and then shuffled them. This is done to facilitate streaming the data directly from the hub!\n\n\t\t\n\t\tLoad… See the full description on the dataset page: https://huggingface.co/datasets/HuggingFaceM4/FineVisionMax.","tags":"[\"Task_categories:image-Text-To-Text\",\"Language:en\",\"Language:zh\",\"Size_categories:10M<n<100M\",\"Format:parquet\",\"Format:optimized-Parquet\"]","license":null,"framework":null,"parameters":null,"downloads":93503,"likes":32,"verified":0,"created_at":"2026-08-11 10:23:34","updated_at":"2026-08-14 06:23:29","source_url":"https://huggingface.co/datasets/HuggingFaceM4/FineVisionMax","source_platform":"huggingface","hf_repo_id":"HuggingFaceM4/FineVisionMax","ollama_name":"","category":"dataset","latest_version":"v1.0.0","version_count":1,"signature_count":1,"risk_level":null,"risk_score":null,"versions":[{"id":796,"model_id":797,"version":"v1.0.0","manifest_hash":"f1021158843b32667920f0299c0e4a1bd315eb4920e81811c6c36862409f4636","file_count":0,"total_size":0,"r2_manifest_key":"manifests/datasets/huggingfacem4--finevisionmax/v1.0.0.json","created_at":"2026-08-11 10:23:34"}],"files":[],"signatures":[{"id":1369,"version_id":796,"signer_did":"did:quantamrkt:registry:shield-v1","algorithm":"ML-DSA-65","signature_hex":"3c9d56269830054e9ee3f43c8726b2389e23d24bee2efedd476ea762b77f9743","attestation_type":"registry","signed_at":"2026-08-11 10:23:34"}],"hndl":null}