{"crate":{"id":"extractous","name":"extractous","updated_at":"2024-12-21T09:19:26.765090Z","versions":[1382253,1343395,1327600,1318981,1274417,1267440,1267405,1267399],"keywords":["parser","pdf","text","tika","unstructured"],"categories":["parsing","text-processing"],"badges":[],"created_at":"2024-09-11T14:06:34.054620Z","downloads":642307,"recent_downloads":289953,"default_version":"0.3.0","num_versions":8,"yanked":false,"max_version":"0.3.0","newest_version":"0.3.0","max_stable_version":"0.3.0","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by the Rust\ncore\n","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","links":{"version_downloads":"/api/v1/crates/extractous/downloads","versions":null,"owners":"/api/v1/crates/extractous/owners","owner_team":"/api/v1/crates/extractous/owner_team","owner_user":"/api/v1/crates/extractous/owner_user","reverse_dependencies":"/api/v1/crates/extractous/reverse_dependencies"},"exact_match":false,"trustpub_only":false},"versions":[{"id":1382253,"crate":"extractous","num":"0.3.0","dl_path":"/api/v1/crates/extractous/0.3.0/download","readme_path":"/api/v1/crates/extractous/0.3.0/readme","updated_at":"2024-12-21T09:19:26.765090Z","created_at":"2024-12-21T09:19:26.765090Z","downloads":634302,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.3.0/dependencies","version_downloads":"/api/v1/crates/extractous/0.3.0/downloads","authors":"/api/v1/crates/extractous/0.3.0/authors"},"crate_size":135794,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-12-21T09:19:26.765090Z"}],"checksum":"082fd3334d09f6722e230d3a824e4ebab34bcf88e8d40b40c0bdb806d436d3f4","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by the Rust\ncore\n","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":485,"comment_lines":112,"files":5},"Rust":{"code_lines":1501,"comment_lines":171,"files":8}},"total_code_lines":1986,"total_comment_lines":283}},{"id":1343395,"crate":"extractous","num":"0.2.0","dl_path":"/api/v1/crates/extractous/0.2.0/download","readme_path":"/api/v1/crates/extractous/0.2.0/readme","updated_at":"2024-11-17T23:28:34.204432Z","created_at":"2024-11-17T23:28:34.204432Z","downloads":1679,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.2.0/dependencies","version_downloads":"/api/v1/crates/extractous/0.2.0/downloads","authors":"/api/v1/crates/extractous/0.2.0/authors"},"crate_size":134000,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-11-17T23:28:34.204432Z"}],"checksum":"cfc454a41e38531aa560abc8d4c3b4c02b94da32013720af3bc9c4e171025501","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by the Rust\ncore","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":379,"comment_lines":110,"files":4},"Rust":{"code_lines":1412,"comment_lines":166,"files":8}},"total_code_lines":1791,"total_comment_lines":276}},{"id":1327600,"crate":"extractous","num":"0.1.7","dl_path":"/api/v1/crates/extractous/0.1.7/download","readme_path":"/api/v1/crates/extractous/0.1.7/readme","updated_at":"2024-11-04T20:06:09.465865Z","created_at":"2024-11-04T20:06:09.465865Z","downloads":1254,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.1.7/dependencies","version_downloads":"/api/v1/crates/extractous/0.1.7/downloads","authors":"/api/v1/crates/extractous/0.1.7/authors"},"crate_size":109870,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-11-04T20:06:09.465865Z"}],"checksum":"5f4b8f148289ba08a3625a3b2fffea9a67e7af018445c1796841d2a9273f73d1","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by the Rust\ncore","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":242,"comment_lines":84,"files":3},"Rust":{"code_lines":1092,"comment_lines":172,"files":8}},"total_code_lines":1334,"total_comment_lines":256}},{"id":1318981,"crate":"extractous","num":"0.1.5","dl_path":"/api/v1/crates/extractous/0.1.5/download","readme_path":"/api/v1/crates/extractous/0.1.5/readme","updated_at":"2024-10-29T11:32:14.554690Z","created_at":"2024-10-29T11:32:14.554690Z","downloads":900,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.1.5/dependencies","version_downloads":"/api/v1/crates/extractous/0.1.5/downloads","authors":"/api/v1/crates/extractous/0.1.5/authors"},"crate_size":107660,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-10-29T11:32:14.554690Z"}],"checksum":"57368a311fcec94fa36fdf644316ac3fdb96fd9c6fdceb63efb981224773f645","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by the Rust\ncore","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":203,"comment_lines":83,"files":3},"Rust":{"code_lines":1065,"comment_lines":191,"files":8}},"total_code_lines":1268,"total_comment_lines":274}},{"id":1274417,"crate":"extractous","num":"0.1.4","dl_path":"/api/v1/crates/extractous/0.1.4/download","readme_path":"/api/v1/crates/extractous/0.1.4/readme","updated_at":"2024-09-18T19:44:38.070551Z","created_at":"2024-09-18T19:44:38.070551Z","downloads":1062,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.1.4/dependencies","version_downloads":"/api/v1/crates/extractous/0.1.4/downloads","authors":"/api/v1/crates/extractous/0.1.4/authors"},"crate_size":104576,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-09-18T19:44:38.070551Z"}],"checksum":"a5dceb9a65df8eed8c7a258fbac415c584c882a71122bc72646fbdacd5292139","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by the Rust\ncore","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":234,"comment_lines":99,"files":4},"Rust":{"code_lines":1021,"comment_lines":177,"files":8}},"total_code_lines":1255,"total_comment_lines":276}},{"id":1267440,"crate":"extractous","num":"0.1.3","dl_path":"/api/v1/crates/extractous/0.1.3/download","readme_path":"/api/v1/crates/extractous/0.1.3/readme","updated_at":"2024-09-11T15:37:21.936676Z","created_at":"2024-09-11T15:37:21.936676Z","downloads":1008,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.1.3/dependencies","version_downloads":"/api/v1/crates/extractous/0.1.3/downloads","authors":"/api/v1/crates/extractous/0.1.3/authors"},"crate_size":103960,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-09-11T15:37:21.936676Z"}],"checksum":"81da6f60cb37f1ec616780af1a29abb3958ca3d82e5e149bf0a0ffa5ddae1b77","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by Rust","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":234,"comment_lines":99,"files":4},"Rust":{"code_lines":1004,"comment_lines":177,"files":8}},"total_code_lines":1238,"total_comment_lines":276}},{"id":1267405,"crate":"extractous","num":"0.1.2","dl_path":"/api/v1/crates/extractous/0.1.2/download","readme_path":"/api/v1/crates/extractous/0.1.2/readme","updated_at":"2024-09-11T14:19:26.244596Z","created_at":"2024-09-11T14:19:26.244596Z","downloads":1166,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.1.2/dependencies","version_downloads":"/api/v1/crates/extractous/0.1.2/downloads","authors":"/api/v1/crates/extractous/0.1.2/authors"},"crate_size":103974,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-09-11T14:19:26.244596Z"}],"checksum":"8f911b3e073d7af20759d18e3e1fe4db0cdc8a169d7d32765c1dad923b329694","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by Rust","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":234,"comment_lines":99,"files":4},"Rust":{"code_lines":1003,"comment_lines":177,"files":8}},"total_code_lines":1237,"total_comment_lines":276}},{"id":1267399,"crate":"extractous","num":"0.1.1","dl_path":"/api/v1/crates/extractous/0.1.1/download","readme_path":"/api/v1/crates/extractous/0.1.1/readme","updated_at":"2024-09-11T14:06:34.054620Z","created_at":"2024-09-11T14:06:34.054620Z","downloads":936,"features":{},"yanked":false,"yank_message":null,"lib_links":null,"license":"Apache-2.0","links":{"dependencies":"/api/v1/crates/extractous/0.1.1/dependencies","version_downloads":"/api/v1/crates/extractous/0.1.1/downloads","authors":"/api/v1/crates/extractous/0.1.1/authors"},"crate_size":103906,"published_by":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"audit_actions":[{"action":"publish","user":{"id":271829,"login":"nmammeri","name":"Nadjib","avatar":"https://avatars.githubusercontent.com/u/3158098?v=4","url":"https://github.com/nmammeri","github_username_matches":true,"created_at":"2012-12-31T09:45:44Z"},"time":"2024-09-11T14:06:34.054620Z"}],"checksum":"27266fe5aca328485b4785cd0620cfa63d8f7bff7a7290697be2d049e30a1591","rust_version":null,"has_lib":true,"bin_names":[],"edition":"2021","description":"Extractous provides a fast and efficient way to extract content from all kind of file formats including PDF, Word, Excel\nCSV, Email etc... Internally it uses a natively compiled Apache Tika for formats are not supported natively by Rust","homepage":"https://extractous.yobix.ai","documentation":null,"repository":"https://github.com/yobix-ai/extractous","trustpub_data":null,"linecounts":{"languages":{"Java":{"code_lines":234,"comment_lines":99,"files":4},"Rust":{"code_lines":1001,"comment_lines":176,"files":8}},"total_code_lines":1235,"total_comment_lines":275}}],"keywords":[{"id":"parser","keyword":"parser","created_at":"2014-11-14T20:00:40.785931Z","crates_cnt":4726},{"id":"pdf","keyword":"pdf","created_at":"2016-07-23T04:09:47.826781Z","crates_cnt":739},{"id":"text","keyword":"text","created_at":"2014-11-14T08:25:07.725999Z","crates_cnt":1158},{"id":"tika","keyword":"tika","created_at":"2024-09-11T14:06:34.054620Z","crates_cnt":1},{"id":"unstructured","keyword":"unstructured","created_at":"2019-05-01T21:05:40.709210Z","crates_cnt":3}],"categories":[{"id":"parsing","category":"Parsing tools","slug":"parsing","description":"Crates to help create parsers of binary and text formats. Format-specific parsers belong in other, more specific categories.","created_at":"2017-01-17T19:13:05.112025Z","crates_cnt":5451},{"id":"text-processing","category":"Text processing","slug":"text-processing","description":"Crates to deal with the complexities of human language when expressed in textual form.","created_at":"2017-01-17T19:13:05.112025Z","crates_cnt":5737}]}