Buckets:
| import copy | |
| import json | |
| from pathlib import Path | |
| import pytest | |
| from castle_pipeline.schema import validate_annotation, validate_audio, apply_review | |
| def annotation(): | |
| path = Path(__file__).resolve().parents[2] / 'annotation_design_v1' / '输出示例_假设场景.json' | |
| return json.loads(path.read_text(encoding='utf-8')) | |
| def test_rejects_out_of_clip_evidence_and_unknown_actor(annotation): | |
| validate_annotation(annotation, 20, {'video_0', 'audio_0'}) | |
| annotation['segments'][0]['evidence'][0]['end_sec'] = 21 | |
| with pytest.raises(ValueError): | |
| validate_annotation(annotation, 20, {'video_0', 'audio_0'}) | |
| annotation['segments'][0]['evidence'][0]['end_sec'] = 4 | |
| annotation['segments'][0]['actor_ids'] = ['person_99'] | |
| with pytest.raises(ValueError): | |
| validate_annotation(annotation, 20, {'video_0', 'audio_0'}) | |
| def test_visual_only_call_cannot_claim_audio_checked_speech(annotation): | |
| with pytest.raises(ValueError): | |
| validate_annotation(annotation, 20, {'video_0'}) | |
| def test_source_modality_cannot_be_fabricated(annotation, source, modality): | |
| annotation['segments'][0]['evidence'][0].update(source_id=source, modality=modality) | |
| with pytest.raises(ValueError): | |
| validate_annotation(annotation, 20, {'video_0', 'audio_0'}) | |
| def test_transcript_only_speech_requires_supplied_transcript(annotation): | |
| annotation['segments'][2]['speech']['verification'] = 'transcript_only' | |
| with pytest.raises(ValueError): | |
| validate_annotation(annotation, 20, {'video_0', 'audio_0'}) | |
| def test_review_matches_ids_and_applies_empty_corrections(annotation): | |
| replacement = copy.deepcopy(annotation['segments'][0]) | |
| replacement['activity'] = None | |
| replacement['details'] = None | |
| replacement['objects'] = [] | |
| review = {'findings': [], 'segment_replacements': [ | |
| {'segment_id': 's001', 'reason': 'The evidence does not establish the activity.', 'replacement': replacement}], | |
| 'initial_environment_replacement': None, 'scene_summary_replacement': None, | |
| 'resegmentation_requests': []} | |
| revised = apply_review(annotation, review, 20, {'video_0', 'audio_0'}) | |
| assert revised['segments'][0]['activity'] is None | |
| assert revised['segments'][0]['objects'] == [] | |
| assert annotation['segments'][0]['activity'] == 'Preparing a drink' | |
| review['segment_replacements'][0]['replacement']['start_sec'] = 1 | |
| with pytest.raises(ValueError): | |
| apply_review(annotation, review, 20, {'video_0', 'audio_0'}) | |
| def test_audio_distinguishes_unintelligible_speech_from_non_speech(): | |
| payload = {'summary': 'A voice and a clatter.', 'utterances': [ | |
| {'start_sec': 1, 'end_sec': 2, 'speaker_id': 'speaker_1', 'source': 'unknown', | |
| 'text': None, 'intelligibility': 'unintelligible'}], | |
| 'sound_events': [{'start_sec': 3, 'end_sec': 4, 'description': 'A clatter.', 'source': 'unknown'}], | |
| 'uncertainties': []} | |
| validate_audio(payload, 5) | |
| payload['utterances'][0]['end_sec'] = None | |
| with pytest.raises(ValueError): | |
| validate_audio(payload, 5) | |
Xet Storage Details
- Size:
- 3.22 kB
- Xet hash:
- e5988004b70ce93622d8deee26d7f651870205af4ee2d409eef5b890225be05a
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.