<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
	<channel>
		<title>Computer Vision (UCB FA24 CS180) on itu</title>
		<link>/series/computer-vision-ucb-fa24-cs180/</link>
		<description>Recent content in Computer Vision (UCB FA24 CS180) on itu</description>
		<generator>Hugo</generator>
		<language>en</language>
		
		
		
			<copyright>&amp;copy; 2024 itsublog. All rights reserved.
</copyright>
		
		
			<lastBuildDate>Sat, 10 Oct 2026 02:49:00 +0900</lastBuildDate>
		
			<atom:link href="/series/computer-vision-ucb-fa24-cs180/index.xml" rel="self" type="application/rss+xml" />
			<item>
				<title>Facial Keypoint Detection by Regression &amp; Pixelwise Classification</title>
				<link>/posts/cs180-pjs/facial-keypoint-detection/</link>
				<pubDate>Wed, 04 Dec 2024 20:39:29 +0800</pubDate>
				<guid>/posts/cs180-pjs/facial-keypoint-detection/</guid>
				<description>&lt;p&gt;The project spec is &lt;a href=&#34;https://web.archive.org/web/20221128011635/https://inst.eecs.berkeley.edu/~cs194-26/fa22/hw/proj5/&#34;&gt;archived here&lt;/a&gt;.&lt;/p&gt;&#xA;&lt;p&gt;You can find the code &lt;a href=&#34;https://github.com/itsuitsuki/facial-keypoint-detection&#34;&gt;here&lt;/a&gt;.&lt;/p&gt;&#xA;&lt;h1 id=&#34;part-1-nose-tip-detection&#34;&gt;Part 1: Nose Tip Detection&lt;/h1&gt;&#xA;&lt;p&gt;For the first part, we use the IMM Face Database for training an initial toy model for nose tip detection. The dataset has 240 facial images of 40 persons and each person has 6 facial images in different viewpoints. All images are annotated with 58 facial keypoints. We use all 6 images of the first 32 persons (index 1-32) as the training set (total 32 x 6 = 192 images) and the images of the remaining 8 persons (index 33-40) (8 * 6 = 48 images) as the validation set.&lt;/p&gt;</description>
			</item>
			<item>
				<title>[Flow] Rectified Flow Explained</title>
				<link>/posts/cs180-pjs/rectified-flow/</link>
				<pubDate>Wed, 20 Nov 2024 20:39:29 +0800</pubDate>
				<guid>/posts/cs180-pjs/rectified-flow/</guid>
				<description>&lt;p&gt;Here is the (simple) explanation for the framework &lt;strong&gt;&lt;a href=&#34;https://arxiv.org/abs/2209.03003&#34;&gt;Rectified Flow&lt;/a&gt;&lt;/strong&gt;.&lt;/p&gt;&#xA;&lt;p&gt;&lt;a href=&#34;https://github.com/itsuitsuki/simple-rectified-flow&#34;&gt;&lt;strong&gt;A simple implementation&lt;/strong&gt;&lt;/a&gt;&lt;/p&gt;&#xA;&lt;h1 id=&#34;overview&#34;&gt;Overview&lt;/h1&gt;&#xA;&lt;p&gt;Rectified Flow (RF) is a generative modeling method, which tries to transport data from source distribution \(\pi_0\) (which corresponds to the pure Gaussian distribution \(\pi_0=N(0,I)\)) and the target distribution \(\pi_1\), which is the distribution of clean images.&lt;/p&gt;&#xA;&lt;p&gt;The overall objective is to &lt;strong&gt;align the velocity estimate&lt;/strong&gt; (using the UNet, denoted as \(v_\theta\) now) &lt;strong&gt;to the actual velocity&lt;/strong&gt; between the source image \(X_0\) and the target image \(X_1\). First, the timesteps here are all normalized between \(t\in[0,1]\), instead of spreading in \(\{0,\cdots, T\}\).&lt;/p&gt;</description>
			</item>
			<item>
				<title>DDPM Sampling, Applications &amp; Training from Scratch</title>
				<link>/posts/cs180-pjs/ddpm-hw/</link>
				<pubDate>Sun, 10 Nov 2024 01:01:49 -0800</pubDate>
				<guid>/posts/cs180-pjs/ddpm-hw/</guid>
				<description>&lt;p&gt;The original project spec at UC Berkeley CS180 is &lt;a href=&#34;https://cal-cs180.github.io/fa24/hw/proj5/index.html&#34;&gt;here&lt;/a&gt;.&lt;/p&gt;&#xA;&lt;p&gt;Mirror of &lt;a href=&#34;https://itsuitsuki.github.io/cs180_pj_webpages/p5_chyzhou&#34;&gt;CS180 Project 5 Showcase Webpage&lt;/a&gt;.&lt;/p&gt;&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/1/9/2/result.png&#34; alt=&#34;&#34; style=&#34;width: 400px;&#34;&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_0_step20.png&#34; alt=&#34;&#34; style=&#34;width: 400px;&#34;&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/2/logo/result.png&#34; alt=&#34;&#34; style=&#34;width: 400px;&#34;&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5b_pics/4/classcond/epoch_20_animation.gif&#34; alt=&#34;Epoch 20 Animation&#34; style=&#34;width: 800px;&#34;&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;h1 id=&#34;part-a&#34;&gt;Part A&lt;/h1&gt;&#xA;&lt;h2 id=&#34;0-setup&#34;&gt;0. Setup&lt;/h2&gt;&#xA;&lt;h3 id=&#34;2-stages&#34;&gt;2 Stages&lt;/h3&gt;&#xA;&lt;p&gt;We first use 3 prompts to let the model generate output images. Here are images and captions displayed below, with different inference steps:&lt;/p&gt;&#xA;&lt;ul&gt;&#xA;&lt;li&gt;5 steps (i.e. &lt;code&gt;num_inference_steps=5&lt;/code&gt;):&#xA;&lt;ul&gt;&#xA;&lt;li&gt;Size: 64px * 64px (Stage 1)&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_0_step5.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;      &lt;figcaption&gt;an oil painting of &lt;br&gt; a snowy mountain village&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_1_step5.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a man wearing a hat&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_2_step5.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a rocket ship&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Size: 256px * 256px (Stage 2)&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_0_step5.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;      &lt;figcaption&gt;an oil painting of a snowy mountain village&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_1_step5.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a man wearing a hat&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_2_step5.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a rocket ship&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;20 steps:&#xA;&lt;ul&gt;&#xA;&lt;li&gt;Stage 1:&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_0_step20.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;      &lt;figcaption&gt;an oil painting of a &lt;br&gt; snowy mountain village&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_1_step20.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a man wearing a hat&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_2_step20.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a rocket ship&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Stage 2:&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_0_step20.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;      &lt;figcaption&gt;an oil painting of a &lt;br&gt; snowy mountain village&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_1_step20.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a man wearing a hat&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_2_step20.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a rocket ship&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;100 steps:&#xA;&lt;ul&gt;&#xA;&lt;li&gt;Stage 1:&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_0_step100.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;      &lt;figcaption&gt;an oil painting of &lt;br&gt; a snowy mountain village&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_1_step100.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a man wearing a hat&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_1_im_2_step100.png&#34; alt=&#34;&#34; style=&#34;width: 128px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a rocket ship&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Stage 2:&#xA;&lt;div style=&#34;display: flex; justify-content: space-around;&#34;&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_0_step100.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;      &lt;figcaption&gt;an oil painting of a &lt;br&gt; snowy mountain village&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_1_step100.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a man wearing a hat&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;  &lt;figure style=&#34;text-align: center; margin: 10px;&#34;&gt;&#xA;    &lt;img src=&#34;p5a_pics/0/stage_2_im_2_step100.png&#34; alt=&#34;&#34; style=&#34;width: 512px;&#34;&gt;&#xA;    &#x9;&lt;figcaption&gt;a rocket ship&lt;/figcaption&gt;&#xA;  &lt;/figure&gt;&#xA;&lt;/div&gt;&#xA;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;h3 id=&#34;reflection-on-the-generation&#34;&gt;Reflection on the generation&lt;/h3&gt;&#xA;&lt;p&gt;We find that for 5 steps, the outputs are not so clear, specifically, the noise added are not removed so completely. We can observe lots of noisy dots in the generated images. The generated feature is also not so clear.&lt;/p&gt;</description>
			</item>
			<item>
				<title>CS180: Project Webpage Catalog</title>
				<link>/posts/cs180-pjs/catalog/</link>
				<pubDate>Fri, 06 Sep 2024 18:33:06 -0700</pubDate>
				<guid>/posts/cs180-pjs/catalog/</guid>
				<description>&lt;ul&gt;&#xA;&lt;li&gt;Project 1: Image Alignment using Multi-scale Pyramids&#xA;&lt;ul&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://cal-cs180.github.io/fa24/hw/proj1/index.html&#34;&gt;Description&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://itsuitsuki.github.io/cs180_pj_webpages/p1&#34;&gt;Project Webpage&lt;/a&gt;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Project 2: Multi-resolution Blending with Laplacian/Gaussian Filters and Stacks&#xA;&lt;ul&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://cal-cs180.github.io/fa24/hw/proj2/index.html&#34;&gt;Description&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://itsuitsuki.github.io/cs180_pj_webpages/p2&#34;&gt;Project Webpage&lt;/a&gt;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Project 3: Face Morphing&#xA;&lt;ul&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://cal-cs180.github.io/fa24/hw/proj3/index.html&#34;&gt;Description&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://itsuitsuki.github.io/cs180_pj_webpages/p3&#34;&gt;Project Webpage&lt;/a&gt;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Project 4: Feature Recognition, Matching and Image Stitching&#xA;&lt;ul&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://cal-cs180.github.io/fa24/hw/proj4/index.html&#34;&gt;Description&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://itsuitsuki.github.io/cs180_pj_webpages/p4&#34;&gt;Project Webpage&lt;/a&gt;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Project 5: Diffusion Models (DDPM, its principles and applications &amp;amp; Rectified Flow)&#xA;&lt;ul&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://cal-cs180.github.io/fa24/hw/proj5/index.html&#34;&gt;Description&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://itsuitsuki.github.io/cs180_pj_webpages/p5_chyzhou&#34;&gt;Project Webpage&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;../ddpm-hw&#34;&gt;Blog Mirror of Project Webpage&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;../rectified-flow&#34;&gt;Rectified Flow&lt;/a&gt;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;li&gt;Final Projects: Collaborated with Junye Wang&#xA;&lt;ul&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://anubisyy.github.io/COMPSCI180_website/Final_Project.html&#34;&gt;Project Webpage Aggregated&lt;/a&gt;&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;../facial-keypoint-detection&#34;&gt;1/3: Facial Keypoint Detection&lt;/a&gt;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;/li&gt;&#xA;&lt;/ul&gt;</description>
			</item>
	</channel>
</rss>
