@@ -104,16 +104,20 @@ def mg_motionvectordata(self: "musicalgestures.MgVideo") -> "MgMotionVectorData"
104104 predicted. Without it a B-frame's vectors point backwards half the time and averaging
105105 them gives roughly nothing.
106106
107- **What that normalisation cannot fix is the reference distance.** `source` records
108- only the direction, plus or minus one, never how many frames away the reference was.
109- An encoder with multiple reference frames will predict some blocks from two or four
110- frames back, and those read as two or four times the per-frame displacement. Measured
111- on a block moving 4 pixels per frame, 53 per cent of vectors came back as 4 and the
112- rest as 8 or 16, with `source` at plus or minus one throughout. So `median_dx` and
113- `median_dy` are medians for a reason and are robust to it; `magnitude` is a sum and is
114- not, and inherits the over-count. Treat `magnitude` as a quantity to correlate against
115- itself over time, which is what it was validated for, rather than as pixels per
116- second.
107+ **The reference distance is corrected from the cadence, with two residuals.**
108+ `source` records only the direction, plus or minus one, never how many frames away
109+ the reference was, so the distance is recovered instead from the picture-type
110+ cadence: past-referencing vectors are divided by the display-order gap to the
111+ previous reference frame, which makes a P-frame after a run of B-frames read
112+ per-frame displacement rather than the whole span's. What the cadence cannot see:
113+ blocks reaching an OLDER reference through multi-reference prediction --- measured
114+ on a block moving 4 pixels per frame, such blocks read as 8 or 16 with `source` at
115+ plus or minus one throughout --- and B-frame vectors pointing at a future
116+ reference, which default filtering excludes. So `median_dx` and `median_dy` are
117+ medians for a reason and are robust to the residuals; `magnitude` is a sum and
118+ inherits what remains of the over-count. Treat `magnitude` as a quantity to
119+ correlate against itself over time, which is what it was validated for, rather
120+ than as pixels per second.
117121
118122 Returns:
119123 MgMotionVectorData: `time` in seconds, `picture_type` as 'I'/'P'/'B',
@@ -152,10 +156,27 @@ def mg_motionvectordata(self: "musicalgestures.MgVideo") -> "MgMotionVectorData"
152156 magnitude : list [float ] = []
153157 med_dx : list [float ] = []
154158 med_dy : list [float ] = []
159+ #: The reference distance, recovered from the cadence. `source` carries only the
160+ #: SIGN of the reference --- measured on a real encode it is plus or minus one
161+ #: throughout, whatever the distance --- so a P-frame that follows a run of
162+ #: B-frames reports its whole multi-frame displacement as one frame's. What IS
163+ #: recoverable while decoding in display order is the gap back to the previous
164+ #: reference frame (I or P), and dividing the past-referencing vectors by that gap
165+ #: corrects the cadence part exactly. Two residuals remain and are documented
166+ #: rather than hidden: blocks reaching an OLDER reference through multi-reference
167+ #: prediction, and B-frame vectors pointing at a future reference --- the toolbox
168+ #: excludes B-frames by default, and their vectors were already the worse signal.
169+ display_idx = - 1
170+ last_ref_idx = None
155171 for frame in container .decode (stream ):
172+ display_idx += 1
173+ kind = _PICTURE_TYPES .get (int (frame .pict_type ), "?" )
174+ past_gap = 1.0 if last_ref_idx is None else max (1.0 , display_idx - last_ref_idx )
175+ if kind in ("I" , "P" ):
176+ last_ref_idx = display_idx
156177 time .append (float (frame .pts * stream .time_base ) if frame .pts is not None
157178 else (time [- 1 ] if time else 0.0 ))
158- kinds .append (_PICTURE_TYPES . get ( int ( frame . pict_type ), "?" ) )
179+ kinds .append (kind )
159180 vectors = frame .side_data .get ("MOTION_VECTORS" )
160181 table = vectors .to_ndarray () if vectors is not None else None
161182 if table is None or len (table ) == 0 :
@@ -165,20 +186,20 @@ def mg_motionvectordata(self: "musicalgestures.MgVideo") -> "MgMotionVectorData"
165186 med_dy .append (0.0 )
166187 continue
167188 scale = np .maximum (table ["motion_scale" ].astype (np .float64 ), 1 )
168- #: Two corrections in one division, and both are needed to get a number that
169- #: means "which way did this move, per frame".
170- #:
171- #: ffmpeg defines the vector as `src = dst + motion / motion_scale`, so it
172- #: points from where the block is now back to where it came from: content
173- #: travelling right carries a NEGATIVE motion_x. And `source` is negative when
174- #: the reference is an earlier frame, positive when it is a later one, with its
175- #: magnitude the distance in frames. Dividing by `source` undoes both --- the
176- #: backwards sense of the vector and the backwards sense of a future reference
177- #: cancel --- and scales a multi-frame prediction down to one frame.
189+ #: Two corrections in one division: ffmpeg defines the vector as
190+ #: `src = dst + motion / motion_scale`, so it points from where the block is
191+ #: now back to where it came from --- content travelling right carries a
192+ #: NEGATIVE motion_x --- and `source` is negative for a past reference,
193+ #: positive for a future one. Dividing by `source` undoes both senses; the
194+ #: cadence gap above then scales the past-referencing multi-frame predictions
195+ #: down to one frame.
178196 source = table ["source" ].astype (np .float64 )
179197 source [source == 0 ] = - 1
180198 dx = table ["motion_x" ].astype (np .float64 ) / scale / source
181199 dy = table ["motion_y" ].astype (np .float64 ) / scale / source
200+ past = source < 0
201+ dx [past ] /= past_gap
202+ dy [past ] /= past_gap
182203 area = table ["w" ].astype (np .float64 ) * table ["h" ]
183204 counts .append (len (table ))
184205 magnitude .append (float ((np .hypot (dx , dy ) * area ).sum ()))
@@ -282,15 +303,25 @@ def motion_vector_grid(filename, deterministic=False, threshold=0.0):
282303 rows = max (1 , - (- height // _GRID ))
283304
284305 previous_time = 0.0
306+ display_idx = - 1
307+ last_ref_idx = None
285308 try :
286309 for frame in container .decode (stream ):
310+ display_idx += 1
311+ kind = _PICTURE_TYPES .get (int (frame .pict_type ), "?" )
312+ #: The cadence gap to the previous reference frame; see motionvectordata
313+ #: for why `source` cannot provide the distance itself.
314+ past_gap = (1.0 if last_ref_idx is None
315+ else max (1.0 , display_idx - last_ref_idx ))
316+ if kind in ("I" , "P" ):
317+ last_ref_idx = display_idx
287318 vx = np .zeros ((rows , cols ), np .float64 )
288319 vy = np .zeros ((rows , cols ), np .float64 )
289320 w = np .zeros ((rows , cols ), np .float64 )
290321 t = (float (frame .pts * stream .time_base )
291322 if frame .pts is not None else previous_time )
292323 previous_time = t
293- is_p = _PICTURE_TYPES . get ( int ( frame . pict_type ), "?" ) == "P"
324+ is_p = kind == "P"
294325 mvs = frame .side_data .get ("MOTION_VECTORS" )
295326 table = mvs .to_ndarray () if mvs is not None else None
296327 if table is not None and len (table ):
@@ -299,6 +330,9 @@ def motion_vector_grid(filename, deterministic=False, threshold=0.0):
299330 source [source == 0 ] = - 1
300331 dx = table ["motion_x" ].astype (np .float64 ) / scale / source
301332 dy = table ["motion_y" ].astype (np .float64 ) / scale / source
333+ past = source < 0
334+ dx [past ] /= past_gap
335+ dy [past ] /= past_gap
302336 area = (table ["w" ].astype (np .float64 ) * table ["h" ])
303337 if threshold > 0 :
304338 keep = np .hypot (dx , dy ) >= threshold
0 commit comments